diff --git a/tools/fasta_tools/fasta_compute_length.py b/tools/fasta_tools/fasta_compute_length.py
index 1c8c72c117e..650deda0b5f 100644
--- a/tools/fasta_tools/fasta_compute_length.py
+++ b/tools/fasta_tools/fasta_compute_length.py
@@ -7,46 +7,38 @@ Return sequences whose lengths are within the range.
import sys, os
-seq_hash = {}
-
-def parse_fasta_format(file_handle):
-
- tmp_title = ''
- tmp_seq = ''
- tmp_seq_count = 0
- for i, each_line in enumerate(file_handle):
- each_line = each_line.rstrip('\r\n')
- if (each_line[0] == '>'):
- if (len(tmp_seq) > 0):
- tmp_seq_count += 1
- seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq
- tmp_title = each_line
- tmp_seq = ''
- else:
- tmp_seq = tmp_seq + each_line
- if (each_line.split()[0].isdigit()):
- tmp_seq = tmp_seq + ' '
- if (len(tmp_seq) > 0):
- seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq
-
- return 0
+assert sys.version_info[:2] >= ( 2, 4 )
def __main__():
-
input_filename = sys.argv[1]
output_filename = sys.argv[2]
+ tmp_title = tmp_seq = ''
+ tmp_seq_count = 0
+ seq_hash = {}
+
+ for i, line in enumerate( file( input_filename ) ):
+ line = line.rstrip( '\r\n' )
+ if not line or line.startswith( '#' ):
+ continue
+ if line[0] == '>':
+ if len( tmp_seq ) > 0:
+ tmp_seq_count += 1
+ seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq
+ tmp_title = line
+ tmp_seq = ''
+ else:
+ tmp_seq = "%s%s" % ( tmp_seq, line )
+ if line.split()[0].isdigit():
+ tmp_seq = "%s " % tmp_seq
+ if len( tmp_seq ) > 0:
+ seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq
- input_handle = open(input_filename, 'r')
- parse_fasta_format(input_handle)
- input_handle.close()
-
- output_handle = open(output_filename, 'w')
title_keys = seq_hash.keys()
title_keys.sort()
- for (i, fasta_title) in title_keys:
- tmp_seq = seq_hash[(i, fasta_title)]
- print >> output_handle, "%s\t%d" %(fasta_title[1:], len(tmp_seq))
+ output_handle = open( output_filename, 'w' )
+ for i, fasta_title in title_keys:
+ tmp_seq = seq_hash[ ( i, fasta_title ) ]
+ output_handle.write( "%s\t%d\n" % ( fasta_title[ 1: ], len( tmp_seq ) ) )
output_handle.close()
- return 0
if __name__ == "__main__" : __main__()
\ No newline at end of file
diff --git a/tools/fasta_tools/fasta_filter_by_length.py b/tools/fasta_tools/fasta_filter_by_length.py
index df62379e6a9..4947e7b333f 100644
--- a/tools/fasta_tools/fasta_filter_by_length.py
+++ b/tools/fasta_tools/fasta_filter_by_length.py
@@ -7,64 +7,69 @@ Return sequences whose lengths are within the range.
import sys, os
-seq_hash = {}
+assert sys.version_info[:2] >= ( 2, 4 )
-def parse_fasta_format(file_handle):
-
- tmp_title = ''
- tmp_seq = ''
- tmp_seq_count = 0
- for i, each_line in enumerate(file_handle):
- each_line = each_line.rstrip('\r\n')
- if (each_line[0] == '>'):
- if (len(tmp_seq) > 0):
- tmp_seq_count += 1
- seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq
- tmp_title = each_line
- tmp_seq = ''
- else:
- tmp_seq = tmp_seq + each_line
- if (each_line.split()[0].isdigit()):
- tmp_seq = tmp_seq + ' '
- if (len(tmp_seq) > 0):
- seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq
-
- return 0
+def stop_err( msg ):
+ sys.stderr.write( msg )
+ sys.exit()
def __main__():
input_filename = sys.argv[1]
- min_length = int(sys.argv[2])
- max_length = int(sys.argv[3])
+ try:
+ min_length = int( sys.argv[2] )
+ except:
+ stop_err( "Minimal length of the return sequence requires a numerical value." )
+ try:
+ max_length = int( sys.argv[3] )
+ except:
+ stop_err( "Maximum length of the return sequence requires a numerical value." )
output_filename = sys.argv[4]
+ tmp_title = tmp_seq = ''
+ tmp_seq_count = 0
+ seq_hash = {}
+
+ for i, line in enumerate( file( input_filename ) ):
+ line = line.rstrip( '\r\n' )
+ if not line or line.startswith( '#' ):
+ continue
+ if line[0] == '>':
+ if len( tmp_seq ) > 0:
+ tmp_seq_count += 1
+ seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq
+ tmp_title = line
+ tmp_seq = ''
+ else:
+ tmp_seq = "%s%s" % ( tmp_seq, line )
+ if line.split()[0].isdigit():
+ tmp_seq = "%s " % tmp_seq
+ if len( tmp_seq ) > 0:
+ seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq
- input_handle = open(input_filename, 'r')
- parse_fasta_format(input_handle)
- input_handle.close()
-
- output_handle = open(output_filename, 'w')
title_keys = seq_hash.keys()
title_keys.sort()
+ output_handle = open( output_filename, 'w' )
at_least_one = 0
- for (i, fasta_title) in title_keys:
- tmp_seq = seq_hash[(i, fasta_title)]
- if (max_length <= 0):
- compare_max_length = len(tmp_seq)+1
+ for i, fasta_title in title_keys:
+ tmp_seq = seq_hash[ ( i, fasta_title ) ]
+ if max_length <= 0:
+ compare_max_length = len( tmp_seq ) + 1
else:
- compare_max_length = max_length
- if (len(tmp_seq) >= min_length and len(tmp_seq) <= compare_max_length):
+ compare_max_length = max_length
+ l = len( tmp_seq )
+ if l >= min_length and l <= compare_max_length:
at_least_one += 1
- print >> output_handle, "%s" %fasta_title
- l = len(tmp_seq)
+ output_handle.write( "%s\n" % fasta_title )
c = 0
s = tmp_seq
while c < l:
b = min( c + 50, l )
- print >> output_handle, s[c:b]
+ output_handle.write( "%s\n" % s[ c:b ] )
c = b
- if at_least_one == 0: print >> sys.stdout, "There is no sequence that falls within your range."
-
output_handle.close()
- return 0
+
+ if at_least_one == 0:
+ print "There is no sequence that falls within your range."
+
if __name__ == "__main__" : __main__()
\ No newline at end of file
diff --git a/tools/fasta_tools/fasta_filter_by_length.xml b/tools/fasta_tools/fasta_filter_by_length.xml
index 322e5ded10c..27200919ab4 100644
--- a/tools/fasta_tools/fasta_filter_by_length.xml
+++ b/tools/fasta_tools/fasta_filter_by_length.xml
@@ -4,7 +4,7 @@
-
+
@@ -21,13 +21,13 @@
.. class:: infomark
-**TIP**. If only want to show sequences longer than a threshold, set *minimal length* to the threshold and leave *maximal length* to zero.
+**TIP**. If only want to show sequences longer than a threshold, set *minimal length* to the threshold and leave *maximum length* to zero.
-----
**What it does**
- This tool accepts two parameters: *minimal length* and *maximal length*, and returns sequences of length within the two thresholds.
+ This tool accepts two parameters: *minimal length* and *maximum length*, and returns sequences of length within the two thresholds.
-----
@@ -44,7 +44,7 @@
>seq4
ATGGAAGC
-- return sequences with length longer than 10bp (set the *minimal length* to 10, and the *maximal length* to 0 (no limitation))::
+- return sequences with length longer than 10bp (set the *minimal length* to 10, and the *maximum length* to 0 (no limitation))::
>seq1
TCATTTAATGAC
diff --git a/tools/fasta_tools/fasta_to_tabular.py b/tools/fasta_tools/fasta_to_tabular.py
index 5a54570f060..26e538f6a71 100644
--- a/tools/fasta_tools/fasta_to_tabular.py
+++ b/tools/fasta_tools/fasta_to_tabular.py
@@ -18,6 +18,8 @@ def __main__():
sequence_count = 0
for i, line in enumerate( open( infile ) ):
line = line.rstrip( '\r\n' )
+ if not line or line.startswith( '#' ):
+ continue
if line.startswith( '>' ):
if sequence:
sequence_count += 1
@@ -25,19 +27,18 @@ def __main__():
title = line
sequence = ''
else:
- if line:
- sequence += line
- if line.split()[0].isdigit():
- sequence += ' '
+ sequence = "%s%s" % ( sequence, line )
+ if line.split()[0].isdigit():
+ sequence += ' '
if sequence:
seq_hash[( sequence_count, title )] = sequence
# return only those lengths are in the range
- out = open( outfile, 'w' )
title_keys = seq_hash.keys()
title_keys.sort()
+ out = open( outfile, 'w' )
for i, fasta_title in title_keys:
sequence = seq_hash[( i, fasta_title )]
- print >> out, "%s\t%s" %( fasta_title, sequence )
+ out.write( "%s\t%s\n" %( fasta_title, sequence ) )
out.close()
if __name__ == "__main__" : __main__()
\ No newline at end of file