mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Various fixes in fasta tools.
This commit is contained in:
@@ -7,46 +7,38 @@ Return sequences whose lengths are within the range.
|
||||
|
||||
import sys, os
|
||||
|
||||
seq_hash = {}
|
||||
|
||||
def parse_fasta_format(file_handle):
|
||||
|
||||
tmp_title = ''
|
||||
tmp_seq = ''
|
||||
tmp_seq_count = 0
|
||||
for i, each_line in enumerate(file_handle):
|
||||
each_line = each_line.rstrip('\r\n')
|
||||
if (each_line[0] == '>'):
|
||||
if (len(tmp_seq) > 0):
|
||||
tmp_seq_count += 1
|
||||
seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq
|
||||
tmp_title = each_line
|
||||
tmp_seq = ''
|
||||
else:
|
||||
tmp_seq = tmp_seq + each_line
|
||||
if (each_line.split()[0].isdigit()):
|
||||
tmp_seq = tmp_seq + ' '
|
||||
if (len(tmp_seq) > 0):
|
||||
seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq
|
||||
|
||||
return 0
|
||||
assert sys.version_info[:2] >= ( 2, 4 )
|
||||
|
||||
def __main__():
|
||||
|
||||
input_filename = sys.argv[1]
|
||||
output_filename = sys.argv[2]
|
||||
tmp_title = tmp_seq = ''
|
||||
tmp_seq_count = 0
|
||||
seq_hash = {}
|
||||
|
||||
for i, line in enumerate( file( input_filename ) ):
|
||||
line = line.rstrip( '\r\n' )
|
||||
if not line or line.startswith( '#' ):
|
||||
continue
|
||||
if line[0] == '>':
|
||||
if len( tmp_seq ) > 0:
|
||||
tmp_seq_count += 1
|
||||
seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq
|
||||
tmp_title = line
|
||||
tmp_seq = ''
|
||||
else:
|
||||
tmp_seq = "%s%s" % ( tmp_seq, line )
|
||||
if line.split()[0].isdigit():
|
||||
tmp_seq = "%s " % tmp_seq
|
||||
if len( tmp_seq ) > 0:
|
||||
seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq
|
||||
|
||||
input_handle = open(input_filename, 'r')
|
||||
parse_fasta_format(input_handle)
|
||||
input_handle.close()
|
||||
|
||||
output_handle = open(output_filename, 'w')
|
||||
title_keys = seq_hash.keys()
|
||||
title_keys.sort()
|
||||
for (i, fasta_title) in title_keys:
|
||||
tmp_seq = seq_hash[(i, fasta_title)]
|
||||
print >> output_handle, "%s\t%d" %(fasta_title[1:], len(tmp_seq))
|
||||
output_handle = open( output_filename, 'w' )
|
||||
for i, fasta_title in title_keys:
|
||||
tmp_seq = seq_hash[ ( i, fasta_title ) ]
|
||||
output_handle.write( "%s\t%d\n" % ( fasta_title[ 1: ], len( tmp_seq ) ) )
|
||||
output_handle.close()
|
||||
return 0
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
@@ -7,64 +7,69 @@ Return sequences whose lengths are within the range.
|
||||
|
||||
import sys, os
|
||||
|
||||
seq_hash = {}
|
||||
assert sys.version_info[:2] >= ( 2, 4 )
|
||||
|
||||
def parse_fasta_format(file_handle):
|
||||
|
||||
tmp_title = ''
|
||||
tmp_seq = ''
|
||||
tmp_seq_count = 0
|
||||
for i, each_line in enumerate(file_handle):
|
||||
each_line = each_line.rstrip('\r\n')
|
||||
if (each_line[0] == '>'):
|
||||
if (len(tmp_seq) > 0):
|
||||
tmp_seq_count += 1
|
||||
seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq
|
||||
tmp_title = each_line
|
||||
tmp_seq = ''
|
||||
else:
|
||||
tmp_seq = tmp_seq + each_line
|
||||
if (each_line.split()[0].isdigit()):
|
||||
tmp_seq = tmp_seq + ' '
|
||||
if (len(tmp_seq) > 0):
|
||||
seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq
|
||||
|
||||
return 0
|
||||
def stop_err( msg ):
|
||||
sys.stderr.write( msg )
|
||||
sys.exit()
|
||||
|
||||
def __main__():
|
||||
|
||||
input_filename = sys.argv[1]
|
||||
min_length = int(sys.argv[2])
|
||||
max_length = int(sys.argv[3])
|
||||
try:
|
||||
min_length = int( sys.argv[2] )
|
||||
except:
|
||||
stop_err( "Minimal length of the return sequence requires a numerical value." )
|
||||
try:
|
||||
max_length = int( sys.argv[3] )
|
||||
except:
|
||||
stop_err( "Maximum length of the return sequence requires a numerical value." )
|
||||
output_filename = sys.argv[4]
|
||||
tmp_title = tmp_seq = ''
|
||||
tmp_seq_count = 0
|
||||
seq_hash = {}
|
||||
|
||||
for i, line in enumerate( file( input_filename ) ):
|
||||
line = line.rstrip( '\r\n' )
|
||||
if not line or line.startswith( '#' ):
|
||||
continue
|
||||
if line[0] == '>':
|
||||
if len( tmp_seq ) > 0:
|
||||
tmp_seq_count += 1
|
||||
seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq
|
||||
tmp_title = line
|
||||
tmp_seq = ''
|
||||
else:
|
||||
tmp_seq = "%s%s" % ( tmp_seq, line )
|
||||
if line.split()[0].isdigit():
|
||||
tmp_seq = "%s " % tmp_seq
|
||||
if len( tmp_seq ) > 0:
|
||||
seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq
|
||||
|
||||
input_handle = open(input_filename, 'r')
|
||||
parse_fasta_format(input_handle)
|
||||
input_handle.close()
|
||||
|
||||
output_handle = open(output_filename, 'w')
|
||||
title_keys = seq_hash.keys()
|
||||
title_keys.sort()
|
||||
output_handle = open( output_filename, 'w' )
|
||||
at_least_one = 0
|
||||
for (i, fasta_title) in title_keys:
|
||||
tmp_seq = seq_hash[(i, fasta_title)]
|
||||
if (max_length <= 0):
|
||||
compare_max_length = len(tmp_seq)+1
|
||||
for i, fasta_title in title_keys:
|
||||
tmp_seq = seq_hash[ ( i, fasta_title ) ]
|
||||
if max_length <= 0:
|
||||
compare_max_length = len( tmp_seq ) + 1
|
||||
else:
|
||||
compare_max_length = max_length
|
||||
if (len(tmp_seq) >= min_length and len(tmp_seq) <= compare_max_length):
|
||||
compare_max_length = max_length
|
||||
l = len( tmp_seq )
|
||||
if l >= min_length and l <= compare_max_length:
|
||||
at_least_one += 1
|
||||
print >> output_handle, "%s" %fasta_title
|
||||
l = len(tmp_seq)
|
||||
output_handle.write( "%s\n" % fasta_title )
|
||||
c = 0
|
||||
s = tmp_seq
|
||||
while c < l:
|
||||
b = min( c + 50, l )
|
||||
print >> output_handle, s[c:b]
|
||||
output_handle.write( "%s\n" % s[ c:b ] )
|
||||
c = b
|
||||
if at_least_one == 0: print >> sys.stdout, "There is no sequence that falls within your range."
|
||||
|
||||
output_handle.close()
|
||||
return 0
|
||||
|
||||
if at_least_one == 0:
|
||||
print "There is no sequence that falls within your range."
|
||||
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
@@ -4,7 +4,7 @@
|
||||
<inputs>
|
||||
<param name="input" type="data" format="fasta" label="Fasta file"/>
|
||||
<param name="min_length" type="integer" size="15" value="0" label="Minimal length of the return sequence" />
|
||||
<param name="max_length" type="integer" size="15" value="0" label="Maximal length of the return sequence" help="no limitation if 0"/>
|
||||
<param name="max_length" type="integer" size="15" value="0" label="Maximum length of the return sequence" help="no limitation if 0"/>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data name="output" format="fasta"/>
|
||||
@@ -21,13 +21,13 @@
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
**TIP**. If only want to show sequences longer than a threshold, set *minimal length* to the threshold and leave *maximal length* to zero.
|
||||
**TIP**. If only want to show sequences longer than a threshold, set *minimal length* to the threshold and leave *maximum length* to zero.
|
||||
|
||||
-----
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool accepts two parameters: *minimal length* and *maximal length*, and returns sequences of length within the two thresholds.
|
||||
This tool accepts two parameters: *minimal length* and *maximum length*, and returns sequences of length within the two thresholds.
|
||||
|
||||
-----
|
||||
|
||||
@@ -44,7 +44,7 @@
|
||||
>seq4
|
||||
ATGGAAGC
|
||||
|
||||
- return sequences with length longer than 10bp (set the *minimal length* to 10, and the *maximal length* to 0 (no limitation))::
|
||||
- return sequences with length longer than 10bp (set the *minimal length* to 10, and the *maximum length* to 0 (no limitation))::
|
||||
|
||||
>seq1
|
||||
TCATTTAATGAC
|
||||
|
||||
@@ -18,6 +18,8 @@ def __main__():
|
||||
sequence_count = 0
|
||||
for i, line in enumerate( open( infile ) ):
|
||||
line = line.rstrip( '\r\n' )
|
||||
if not line or line.startswith( '#' ):
|
||||
continue
|
||||
if line.startswith( '>' ):
|
||||
if sequence:
|
||||
sequence_count += 1
|
||||
@@ -25,19 +27,18 @@ def __main__():
|
||||
title = line
|
||||
sequence = ''
|
||||
else:
|
||||
if line:
|
||||
sequence += line
|
||||
if line.split()[0].isdigit():
|
||||
sequence += ' '
|
||||
sequence = "%s%s" % ( sequence, line )
|
||||
if line.split()[0].isdigit():
|
||||
sequence += ' '
|
||||
if sequence:
|
||||
seq_hash[( sequence_count, title )] = sequence
|
||||
# return only those lengths are in the range
|
||||
out = open( outfile, 'w' )
|
||||
title_keys = seq_hash.keys()
|
||||
title_keys.sort()
|
||||
out = open( outfile, 'w' )
|
||||
for i, fasta_title in title_keys:
|
||||
sequence = seq_hash[( i, fasta_title )]
|
||||
print >> out, "%s\t%s" %( fasta_title, sequence )
|
||||
out.write( "%s\t%s\n" %( fasta_title, sequence ) )
|
||||
out.close()
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
Reference in New Issue
Block a user