From 498fbac868b1e20178fb89d9141edb9b59d48bf8 Mon Sep 17 00:00:00 2001 From: Wen-Yu Chung Date: Mon, 19 Jan 2009 15:47:02 -0500 Subject: [PATCH] remove the usage of hast tables in the fasta-tools --- tools/fasta_tools/fasta_compute_length.py | 50 +++++++-------- tools/fasta_tools/fasta_filter_by_length.py | 69 ++++++++++++--------- tools/fasta_tools/fasta_to_tabular.py | 39 +++++------- 3 files changed, 80 insertions(+), 78 deletions(-) diff --git a/tools/fasta_tools/fasta_compute_length.py b/tools/fasta_tools/fasta_compute_length.py index 3893072844b..6086ac2716a 100644 --- a/tools/fasta_tools/fasta_compute_length.py +++ b/tools/fasta_tools/fasta_compute_length.py @@ -1,8 +1,8 @@ #! /usr/bin/python """ -Input: fasta, minimal length, maximal length -Output: fasta -Return sequences whose lengths are within the range. +Input: fasta, int +Output: tabular +Return titles with lengths of corresponding seq """ import sys, os @@ -10,41 +10,37 @@ import sys, os assert sys.version_info[:2] >= ( 2, 4 ) def __main__(): - input_filename = sys.argv[1] - output_filename = sys.argv[2] + + infile = sys.argv[1] + outfile = sys.argv[2] keep_first = int( sys.argv[3] ) - tmp_title = tmp_seq = '' - tmp_seq_count = 0 - seq_hash = {} + + fasta_title = fasta_seq = '' + # number of char to keep in the title if keep_first == 0: keep_first = None else: keep_first += 1 - for i, line in enumerate( file( input_filename ) ): + out = open(outfile, 'w') + + for i, line in enumerate( file( infile ) ): line = line.rstrip( '\r\n' ) if not line or line.startswith( '#' ): continue if line[0] == '>': - if len( tmp_seq ) > 0: - tmp_seq_count += 1 - seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq - tmp_title = line - tmp_seq = '' + if len( fasta_seq ) > 0 : + out.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( fasta_seq ) ) ) + fasta_title = line + fasta_seq = '' else: - tmp_seq = "%s%s" % ( tmp_seq, line ) - if line.split() and line.split()[0].isdigit(): - tmp_seq = "%s " % tmp_seq - if len( tmp_seq ) > 0: - seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq - - title_keys = seq_hash.keys() - title_keys.sort() - output_handle = open( output_filename, 'w' ) - for i, fasta_title in title_keys: - tmp_seq = seq_hash[ ( i, fasta_title ) ] - output_handle.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( tmp_seq ) ) ) - output_handle.close() + fasta_seq = "%s%s" % ( fasta_seq, line ) + + # check the last sequence + if len( fasta_seq ) > 0: + out.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( fasta_seq ) ) ) + + out.close() if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/tools/fasta_tools/fasta_filter_by_length.py b/tools/fasta_tools/fasta_filter_by_length.py index 4947e7b333f..9906300e7ef 100644 --- a/tools/fasta_tools/fasta_filter_by_length.py +++ b/tools/fasta_tools/fasta_filter_by_length.py @@ -15,7 +15,7 @@ def stop_err( msg ): def __main__(): - input_filename = sys.argv[1] + infile = sys.argv[1] try: min_length = int( sys.argv[2] ) except: @@ -24,49 +24,62 @@ def __main__(): max_length = int( sys.argv[3] ) except: stop_err( "Maximum length of the return sequence requires a numerical value." ) - output_filename = sys.argv[4] - tmp_title = tmp_seq = '' - tmp_seq_count = 0 - seq_hash = {} + outfile = sys.argv[4] + fasta_title = fasta_seq = '' + at_least_one = 0 + + out = open( outfile, 'w' ) - for i, line in enumerate( file( input_filename ) ): + for i, line in enumerate( file( infile ) ): line = line.rstrip( '\r\n' ) if not line or line.startswith( '#' ): continue + if line[0] == '>': - if len( tmp_seq ) > 0: - tmp_seq_count += 1 - seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq - tmp_title = line - tmp_seq = '' + if len( fasta_seq ) > 0: + + if max_length <= 0: + compare_max_length = len( fasta_seq ) + 1 + else: + compare_max_length = max_length + + l = len( fasta_seq ) + + if l >= min_length and l <= compare_max_length: + at_least_one += 1 + out.write( "%s\n" % fasta_title ) + c = 0 + s = fasta_seq + while c < l: + b = min( c + 50, l ) + out.write( "%s\n" % s[ c:b ] ) + c = b + + fasta_title = line + fasta_seq = '' else: - tmp_seq = "%s%s" % ( tmp_seq, line ) - if line.split()[0].isdigit(): - tmp_seq = "%s " % tmp_seq - if len( tmp_seq ) > 0: - seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq + fasta_seq = "%s%s" % ( fasta_seq, line ) - title_keys = seq_hash.keys() - title_keys.sort() - output_handle = open( output_filename, 'w' ) - at_least_one = 0 - for i, fasta_title in title_keys: - tmp_seq = seq_hash[ ( i, fasta_title ) ] + if len( fasta_seq ) > 0: + if max_length <= 0: - compare_max_length = len( tmp_seq ) + 1 + compare_max_length = len( fasta_seq ) + 1 else: compare_max_length = max_length - l = len( tmp_seq ) + + l = len( fasta_seq ) + if l >= min_length and l <= compare_max_length: at_least_one += 1 - output_handle.write( "%s\n" % fasta_title ) + out.write( "%s\n" % fasta_title ) c = 0 - s = tmp_seq + s = fasta_seq while c < l: b = min( c + 50, l ) - output_handle.write( "%s\n" % s[ c:b ] ) + out.write( "%s\n" % s[ c:b ] ) c = b - output_handle.close() + + out.close() if at_least_one == 0: print "There is no sequence that falls within your range." diff --git a/tools/fasta_tools/fasta_to_tabular.py b/tools/fasta_tools/fasta_to_tabular.py index 42e79e63e2a..8a18d0c69c4 100644 --- a/tools/fasta_tools/fasta_to_tabular.py +++ b/tools/fasta_tools/fasta_to_tabular.py @@ -1,9 +1,9 @@ #! /usr/bin/python # This code exists in 2 places: ~/datatypes/converters and ~/tools/fasta_tools """ -Input: fasta, minimal length, maximal length -Output: fasta -Return sequences whose lengths are within the range. +Input: fasta, int +Output: tabular +format convert: fasta to tabular """ import sys, os @@ -14,38 +14,31 @@ def __main__(): infile = sys.argv[1] outfile = sys.argv[2] keep_first = int( sys.argv[3] ) - title = '' - sequence = '' - sequence_count = 0 + fasta_title = fasta_seq = '' if keep_first == 0: keep_first = None else: keep_first += 1 + out = open( outfile, 'w' ) + for i, line in enumerate( open( infile ) ): line = line.rstrip( '\r\n' ) if not line or line.startswith( '#' ): continue if line.startswith( '>' ): - if sequence: - sequence_count += 1 - seq_hash[( sequence_count, title )] = sequence - title = line - sequence = '' + if fasta_seq: + out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], fasta_seq ) ) + fasta_title = line + fasta_seq = '' else: - sequence = "%s%s" % ( sequence, line ) - if line.split() and line.split()[0].isdigit(): - sequence += ' ' - if sequence: - seq_hash[( sequence_count, title )] = sequence - # return only those lengths are in the range - title_keys = seq_hash.keys() - title_keys.sort() - out = open( outfile, 'w' ) - for i, fasta_title in title_keys: - sequence = seq_hash[( i, fasta_title )] - out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], sequence ) ) + if line: + fasta_seq = "%s%s" % ( fasta_seq, line ) + + if fasta_seq: + out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], fasta_seq ) ) + out.close() if __name__ == "__main__" : __main__() \ No newline at end of file