diff --git a/tools/stats/aggregate_binned_scores_in_intervals.xml b/tools/stats/aggregate_binned_scores_in_intervals.xml index e930e7e8867..221c551d988 100644 --- a/tools/stats/aggregate_binned_scores_in_intervals.xml +++ b/tools/stats/aggregate_binned_scores_in_intervals.xml @@ -1,16 +1,35 @@ - + Appends the average, min, max of datapoints per interval - aggregate_scores_in_intervals.py $datasets $input1 $input1_chromCol $input1_startCol $input1_endCol $out_file1 -b + + #if $score_source_type.score_source == "user":#aggregate_scores_in_intervals.py $score_source_type.input2 $input1 $input1_chromCol $input1_startCol $input1_endCol $out_file1 + #else:#aggregate_scores_in_intervals.py $score_source_type.datasets $input1 $input1_chromCol $input1_startCol $input1_endCol $out_file1 -b + #end if + - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + @@ -18,20 +37,28 @@ + + + + + + + + .. class:: warningmark -This tool currently only works with data from genome builds hg16, hg17 or hg18. +This tool currently only has cached data for genome builds hg16, hg17 and hg18. However, you may use your own data point (wiggle) data, such as is available from UCSC. If you are trying to use your own data point file and it is not appearing as an option, make sure that the builds for your history items are the same. .. class:: warningmark diff --git a/tools/stats/aggregate_binned_scores_in_intervals.xml.old b/tools/stats/aggregate_binned_scores_in_intervals.xml.old new file mode 100644 index 00000000000..e930e7e8867 --- /dev/null +++ b/tools/stats/aggregate_binned_scores_in_intervals.xml.old @@ -0,0 +1,86 @@ + + Appends the average, min, max of datapoints per interval + aggregate_scores_in_intervals.py $datasets $input1 $input1_chromCol $input1_startCol $input1_endCol $out_file1 -b + + + + + + + + + + + + + + + + + + + + + + + + + + + + +.. class:: warningmark + +This tool currently only works with data from genome builds hg16, hg17 or hg18. + +.. class:: warningmark + +This tool assumes that the input dataset is in interval format and contains at least a chrom column, a start column and an end column. These 3 columns can be dispersed throughout any number of other data columns. + +----- + +.. class:: infomark + +**TIP:** Computing summary information may throw exceptions if the data type (e.g., string, integer) in every line of the columns is not appropriate for the computation (e.g., attempting numerical calculations on strings). If an exception is thrown when computing summary information for a line, that line is skipped as invalid for the computation. The number of invalid skipped lines is documented in the resulting history item as a "Data issue". + +----- + +**Syntax** + +This tool appends columns of summary information for each interval matched against a selected dataset. For each interval, the average, minimum and maximum for the data falling within the interval is computed. + +- Several quantitative scores are provided for the ENCODE regions. + + - Various Scores + - Regulatory Potential + - Neutral rate (Ancestral Repeats) + - GC fraction + - Conservation Scores + - PhastCons + - binCons + - GERP + +----- + +**Example** + +If your original data has the following format: + ++------+-----+-----+---+------+ +|other1|chrom|start|end|other2| ++------+-----+-----+---+------+ + +and you choose to aggregate phastCons scores, your output will look like this: + ++------+-----+-----+---+------+---+---+---+ +|other1|chrom|start|end|other2|avg|min|max| ++------+-----+-----+---+------+---+---+---+ + +where: + +* **avg** - average phastCons score for each region +* **min** - minimum phastCons score for each region +* **max** - maximum phastCons score for each region + + + diff --git a/tools/stats/aggregate_scores_in_intervals.py b/tools/stats/aggregate_scores_in_intervals.py index d07850df507..49514c33f80 100755 --- a/tools/stats/aggregate_scores_in_intervals.py +++ b/tools/stats/aggregate_scores_in_intervals.py @@ -146,13 +146,8 @@ def main(): min_score = 100000000 max_score = -100000000 for j in range( start, stop ): - valid2 = True if chrom in scores_by_chrom: try: - scores_by_chrom[chrom][j] - except: - valid2 = False - if valid2: # Skip if base is masked if masks and chrom in masks: if masks[chrom][j]: @@ -164,6 +159,8 @@ def main(): count += 1 max_score = max( score, max_score ) min_score = min( score, min_score ) + except: + continue if count > 0: avg = total/count else: