diff --git a/tools/fasta_tools/fasta_compute_length.py b/tools/fasta_tools/fasta_compute_length.py index 1c8c72c117e..650deda0b5f 100644 --- a/tools/fasta_tools/fasta_compute_length.py +++ b/tools/fasta_tools/fasta_compute_length.py @@ -7,46 +7,38 @@ Return sequences whose lengths are within the range. import sys, os -seq_hash = {} - -def parse_fasta_format(file_handle): - - tmp_title = '' - tmp_seq = '' - tmp_seq_count = 0 - for i, each_line in enumerate(file_handle): - each_line = each_line.rstrip('\r\n') - if (each_line[0] == '>'): - if (len(tmp_seq) > 0): - tmp_seq_count += 1 - seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq - tmp_title = each_line - tmp_seq = '' - else: - tmp_seq = tmp_seq + each_line - if (each_line.split()[0].isdigit()): - tmp_seq = tmp_seq + ' ' - if (len(tmp_seq) > 0): - seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq - - return 0 +assert sys.version_info[:2] >= ( 2, 4 ) def __main__(): - input_filename = sys.argv[1] output_filename = sys.argv[2] + tmp_title = tmp_seq = '' + tmp_seq_count = 0 + seq_hash = {} + + for i, line in enumerate( file( input_filename ) ): + line = line.rstrip( '\r\n' ) + if not line or line.startswith( '#' ): + continue + if line[0] == '>': + if len( tmp_seq ) > 0: + tmp_seq_count += 1 + seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq + tmp_title = line + tmp_seq = '' + else: + tmp_seq = "%s%s" % ( tmp_seq, line ) + if line.split()[0].isdigit(): + tmp_seq = "%s " % tmp_seq + if len( tmp_seq ) > 0: + seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq - input_handle = open(input_filename, 'r') - parse_fasta_format(input_handle) - input_handle.close() - - output_handle = open(output_filename, 'w') title_keys = seq_hash.keys() title_keys.sort() - for (i, fasta_title) in title_keys: - tmp_seq = seq_hash[(i, fasta_title)] - print >> output_handle, "%s\t%d" %(fasta_title[1:], len(tmp_seq)) + output_handle = open( output_filename, 'w' ) + for i, fasta_title in title_keys: + tmp_seq = seq_hash[ ( i, fasta_title ) ] + output_handle.write( "%s\t%d\n" % ( fasta_title[ 1: ], len( tmp_seq ) ) ) output_handle.close() - return 0 if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/tools/fasta_tools/fasta_filter_by_length.py b/tools/fasta_tools/fasta_filter_by_length.py index df62379e6a9..4947e7b333f 100644 --- a/tools/fasta_tools/fasta_filter_by_length.py +++ b/tools/fasta_tools/fasta_filter_by_length.py @@ -7,64 +7,69 @@ Return sequences whose lengths are within the range. import sys, os -seq_hash = {} +assert sys.version_info[:2] >= ( 2, 4 ) -def parse_fasta_format(file_handle): - - tmp_title = '' - tmp_seq = '' - tmp_seq_count = 0 - for i, each_line in enumerate(file_handle): - each_line = each_line.rstrip('\r\n') - if (each_line[0] == '>'): - if (len(tmp_seq) > 0): - tmp_seq_count += 1 - seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq - tmp_title = each_line - tmp_seq = '' - else: - tmp_seq = tmp_seq + each_line - if (each_line.split()[0].isdigit()): - tmp_seq = tmp_seq + ' ' - if (len(tmp_seq) > 0): - seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq - - return 0 +def stop_err( msg ): + sys.stderr.write( msg ) + sys.exit() def __main__(): input_filename = sys.argv[1] - min_length = int(sys.argv[2]) - max_length = int(sys.argv[3]) + try: + min_length = int( sys.argv[2] ) + except: + stop_err( "Minimal length of the return sequence requires a numerical value." ) + try: + max_length = int( sys.argv[3] ) + except: + stop_err( "Maximum length of the return sequence requires a numerical value." ) output_filename = sys.argv[4] + tmp_title = tmp_seq = '' + tmp_seq_count = 0 + seq_hash = {} + + for i, line in enumerate( file( input_filename ) ): + line = line.rstrip( '\r\n' ) + if not line or line.startswith( '#' ): + continue + if line[0] == '>': + if len( tmp_seq ) > 0: + tmp_seq_count += 1 + seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq + tmp_title = line + tmp_seq = '' + else: + tmp_seq = "%s%s" % ( tmp_seq, line ) + if line.split()[0].isdigit(): + tmp_seq = "%s " % tmp_seq + if len( tmp_seq ) > 0: + seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq - input_handle = open(input_filename, 'r') - parse_fasta_format(input_handle) - input_handle.close() - - output_handle = open(output_filename, 'w') title_keys = seq_hash.keys() title_keys.sort() + output_handle = open( output_filename, 'w' ) at_least_one = 0 - for (i, fasta_title) in title_keys: - tmp_seq = seq_hash[(i, fasta_title)] - if (max_length <= 0): - compare_max_length = len(tmp_seq)+1 + for i, fasta_title in title_keys: + tmp_seq = seq_hash[ ( i, fasta_title ) ] + if max_length <= 0: + compare_max_length = len( tmp_seq ) + 1 else: - compare_max_length = max_length - if (len(tmp_seq) >= min_length and len(tmp_seq) <= compare_max_length): + compare_max_length = max_length + l = len( tmp_seq ) + if l >= min_length and l <= compare_max_length: at_least_one += 1 - print >> output_handle, "%s" %fasta_title - l = len(tmp_seq) + output_handle.write( "%s\n" % fasta_title ) c = 0 s = tmp_seq while c < l: b = min( c + 50, l ) - print >> output_handle, s[c:b] + output_handle.write( "%s\n" % s[ c:b ] ) c = b - if at_least_one == 0: print >> sys.stdout, "There is no sequence that falls within your range." - output_handle.close() - return 0 + + if at_least_one == 0: + print "There is no sequence that falls within your range." + if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/tools/fasta_tools/fasta_filter_by_length.xml b/tools/fasta_tools/fasta_filter_by_length.xml index 322e5ded10c..27200919ab4 100644 --- a/tools/fasta_tools/fasta_filter_by_length.xml +++ b/tools/fasta_tools/fasta_filter_by_length.xml @@ -4,7 +4,7 @@ - + @@ -21,13 +21,13 @@ .. class:: infomark -**TIP**. If only want to show sequences longer than a threshold, set *minimal length* to the threshold and leave *maximal length* to zero. +**TIP**. If only want to show sequences longer than a threshold, set *minimal length* to the threshold and leave *maximum length* to zero. ----- **What it does** - This tool accepts two parameters: *minimal length* and *maximal length*, and returns sequences of length within the two thresholds. + This tool accepts two parameters: *minimal length* and *maximum length*, and returns sequences of length within the two thresholds. ----- @@ -44,7 +44,7 @@ >seq4 ATGGAAGC -- return sequences with length longer than 10bp (set the *minimal length* to 10, and the *maximal length* to 0 (no limitation)):: +- return sequences with length longer than 10bp (set the *minimal length* to 10, and the *maximum length* to 0 (no limitation)):: >seq1 TCATTTAATGAC diff --git a/tools/fasta_tools/fasta_to_tabular.py b/tools/fasta_tools/fasta_to_tabular.py index 5a54570f060..26e538f6a71 100644 --- a/tools/fasta_tools/fasta_to_tabular.py +++ b/tools/fasta_tools/fasta_to_tabular.py @@ -18,6 +18,8 @@ def __main__(): sequence_count = 0 for i, line in enumerate( open( infile ) ): line = line.rstrip( '\r\n' ) + if not line or line.startswith( '#' ): + continue if line.startswith( '>' ): if sequence: sequence_count += 1 @@ -25,19 +27,18 @@ def __main__(): title = line sequence = '' else: - if line: - sequence += line - if line.split()[0].isdigit(): - sequence += ' ' + sequence = "%s%s" % ( sequence, line ) + if line.split()[0].isdigit(): + sequence += ' ' if sequence: seq_hash[( sequence_count, title )] = sequence # return only those lengths are in the range - out = open( outfile, 'w' ) title_keys = seq_hash.keys() title_keys.sort() + out = open( outfile, 'w' ) for i, fasta_title in title_keys: sequence = seq_hash[( i, fasta_title )] - print >> out, "%s\t%s" %( fasta_title, sequence ) + out.write( "%s\t%s\n" %( fasta_title, sequence ) ) out.close() if __name__ == "__main__" : __main__() \ No newline at end of file