diff --git a/tools/encode/split_by_partitions.py b/tools/encode/split_by_partitions.py index 831b53002e3..f7ad1870d7e 100755 --- a/tools/encode/split_by_partitions.py +++ b/tools/encode/split_by_partitions.py @@ -112,7 +112,7 @@ def main(): in_file.close() if warnings: - warn_msg = "Total of %d warnings, 1st is: " % len( warnings ) + warn_msg = "This tool is useful on ENCODE regions only, %d warnings, 1st is: " % len( warnings ) warn_msg += warnings[0] print warn_msg if skipped_lines: diff --git a/tools/extract/extract_genomic_dna.xml b/tools/extract/extract_genomic_dna.xml index de8f94763bf..8ec4b813242 100644 --- a/tools/extract/extract_genomic_dna.xml +++ b/tools/extract/extract_genomic_dna.xml @@ -1,4 +1,4 @@ - + using coordinates from assembled/unassebmled genomes extract_genomic_dna.py $input $out_file1 $input_chromCol $input_startCol $input_endCol $input_strandCol $dbkey $out_format @@ -35,7 +35,7 @@ .. class:: warningmark -This tool requires tabular formatted data! +This tool requires tabular formatted data. If your data is not TAB delimited, use *Edit Queries->Convert characters*. .. class:: warningmark diff --git a/tools/new_operations/flanking_features.py b/tools/new_operations/flanking_features.py index 20c9ca57d5f..dca4237ec4c 100644 --- a/tools/new_operations/flanking_features.py +++ b/tools/new_operations/flanking_features.py @@ -20,7 +20,6 @@ def stop_err( msg ): def main(): infile1_includes_strand = False - # Parsing Command Line here options, args = doc_optparse.parse( __doc__ ) try: @@ -81,10 +80,12 @@ def main(): if line and not line.startswith( '#' ): try: elems = line.split( '\t' ) - #if the start and/or end columns are not numbers, skip that line. chr = elems[chr_col_1] start = int( elems[start_col_1] ) end = int( elems[end_col_1] ) + if infile1_includes_strand: + strand = elems[strand_col_1] + assert strand in ['+', '-'] except: skipped_lines += 1 if not invalid_line: @@ -92,27 +93,13 @@ def main(): invalid_line = line continue - if infile1_includes_strand: - #Strand column is defined - try: - strand = elems[strand_col_1] - #if the stand value is not + or -, skip that line. - assert strand in ['+', '-'] - except: - skipped_lines += 1 - if not invalid_line: - first_invalid_line = i + 1 - invalid_line = line - continue - if direction == 'Upstream' or direction == 'Both': for fline in file( tmp_file_up ): fline = fline.rstrip( '\r\n' ) if fline and not fline.startswith( '#' ): try: felems = fline.split( '\t' ) - fchr = felems[chr_col_2] - if fchr != chr: + if chr != felems[chr_col_2]: continue if infile1_includes_strand: try: @@ -125,11 +112,11 @@ def main(): continue fstart = int( felems[start_col_2] ) fend = int( felems[end_col_2] ) - if strand == '+'and fend < start: + if strand == '+' and fend < start: #Highest feature end value encountered i.e. the closest upstream feature found fo.write( "%s\t%s\n" % ( line, fline ) ) break - elif strand == '-'and fstart > end: + elif strand == '-' and fstart > end: #Lowest feature start value encountered i.e. the closest upstream feature found fo.write( "%s\t%s\n" % ( line, fline ) ) break @@ -141,8 +128,7 @@ def main(): if fline and not fline.startswith( '#' ): try: felems = fline.split( '\t' ) - fchr = felems[chr_col_2] - if fchr != chr: + if chr != felems[chr_col_2]: continue if infile1_includes_strand: try: