Fixes for tabular_to_fasta and fasta_to_tabular tools - requires update to datatype_converters_conf.xml

- Use ColumnListParameter rather than TextToolParameter
- Use better naming convertions for code and test-data files
- Added fasta_to_tabular converter to datatype_converters_conf.xml
- Fixed tool-conf so the converters are displayed in the Converters tool section
- Other miscellaneous cleanup.
This commit is contained in:
Greg Von Kuster
2008-01-28 21:25:33 +00:00
parent 9d7dff5c2a
commit 32ccd3570c
11 changed files with 175 additions and 144 deletions
+3 -2
View File
@@ -1,8 +1,9 @@
<?xml version="1.0"?>
<converters>
<converter file="interval_to_bed_converter.xml" source_datatype="interval" target_datatype="bed"/>
<converter file="maf_to_interval_converter.xml" source_datatype="maf" target_datatype="interval"/>
<converter file="bed_to_gff_converter.xml" source_datatype="bed" target_datatype="gff"/>
<converter file="fasta_to_tabular_converter.xml" source_datatype="fasta" target_datatype="tabular"/>
<converter file="gff_to_bed_converter.xml" source_datatype="gff" target_datatype="bed"/>
<converter file="interval_to_bed_converter.xml" source_datatype="interval" target_datatype="bed"/>
<converter file="maf_to_fasta_converter.xml" source_datatype="maf" target_datatype="fasta"/>
<converter file="maf_to_interval_converter.xml" source_datatype="maf" target_datatype="interval"/>
</converters>
@@ -0,0 +1,43 @@
#! /usr/bin/python
# This code exists in 2 places: ~/datatypes/converters and ~/tools/fasta_tools
"""
Input: fasta, minimal length, maximal length
Output: fasta
Return sequences whose lengths are within the range.
Wen-Yu Chung
"""
import sys, os
seq_hash = {}
def __main__():
infile = sys.argv[1]
outfile = sys.argv[2]
title = ''
sequence = ''
sequence_count = 0
for i, line in enumerate( open( infile ) ):
line = line.rstrip( '\r\n' )
if line.startswith( '>' ):
if sequence:
sequence_count += 1
seq_hash[( sequence_count, title )] = sequence
title = line
sequence = ''
else:
sequence += line
if line.split()[0].isdigit():
sequence += ' '
if sequence:
seq_hash[( sequence_count, title )] = sequence
# return only those lengths are in the range
out = open( outfile, 'w' )
title_keys = seq_hash.keys()
title_keys.sort()
for i, fasta_title in title_keys:
sequence = seq_hash[( i, fasta_title )]
print >> out, "%s\t%s" %( fasta_title, sequence )
out.close()
if __name__ == "__main__" : __main__()
@@ -0,0 +1,13 @@
<tool id="CONVERTER_fasta_to_tabular" name="Convert FASTA to Tabular" version="1.0.0">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<command interpreter="python">fasta_to_tabular_converter.py $input $output</command>
<inputs>
<param name="input" type="data" format="fasta" label="Fasta file"/>
</inputs>
<outputs>
<data name="output" format="tabular"/>
</outputs>
<help>
</help>
</tool>
+3
View File
@@ -69,8 +69,11 @@
<tool file="patmat/findcluster_mysql.xml" />
</section> -->
<section name="Fetch Sequences" id="fetchSeq">
<!--
<tool file="extract/fasta-subseq-wrapper.xml" />
<tool file="extract/twoBitToFa_wrapper.xml" />
-->
<tool file="extract/extract_genomic_dna.xml" />
</section>
<section name="Fetch Alignments" id="fetchAlign">
<tool file="maf/interval2maf_pairwise.xml" />
+7 -7
View File
@@ -55,14 +55,16 @@
<tool file="stats/grouping.xml" />
</section>
<section name="Convert Formats" id="convert">
<tool file="maf/maf_to_fasta.xml" />
<tool file="maf/maf_to_bed.xml" />
<tool file="filters/gff2bed.xml" />
<tool file="filters/bed2gff.xml" />
<tool file="filters/axt_to_fasta.xml" />
<tool file="filters/axt_to_concat_fasta.xml" />
<tool file="filters/axt_to_fasta.xml" />
<tool file="filters/axt_to_lav.xml" />
<tool file="filters/bed2gff.xml" />
<tool file="fasta_tools/fasta_to_tabular.xml" />
<tool file="filters/gff2bed.xml" />
<tool file="filters/lav_to_bed.xml" />
<tool file="maf/maf_to_bed.xml" />
<tool file="maf/maf_to_fasta.xml" />
<tool file="fasta_tools/tabular_to_fasta.xml" />
</section>
<section name="Extract Features" id="features">
<tool file="filters/ucsc_gene_bed_to_exon_bed.xml" />
@@ -303,8 +305,6 @@
<section name="Fasta Tools" id="fasta_tools">
<tool file="fasta_tools/fasta_tool_compute_length.xml" />
<tool file="fasta_tools/fasta_tool_filter_by_length.xml" />
<tool file="fasta_tools/fasta_tool_fasta2tab.xml" />
<tool file="fasta_tools/fasta_tool_tab2fasta.xml" />
</section>
<section name="Metagenomics" id="metagenomics">
<tool file="metag_tools/short_reads_trim_seq.xml" />
+43
View File
@@ -0,0 +1,43 @@
#! /usr/bin/python
# This code exists in 2 places: ~/datatypes/converters and ~/tools/fasta_tools
"""
Input: fasta, minimal length, maximal length
Output: fasta
Return sequences whose lengths are within the range.
Wen-Yu Chung
"""
import sys, os
seq_hash = {}
def __main__():
infile = sys.argv[1]
outfile = sys.argv[2]
title = ''
sequence = ''
sequence_count = 0
for i, line in enumerate( open( infile ) ):
line = line.rstrip( '\r\n' )
if line.startswith( '>' ):
if sequence:
sequence_count += 1
seq_hash[( sequence_count, title )] = sequence
title = line
sequence = ''
else:
sequence += line
if line.split()[0].isdigit():
sequence += ' '
if sequence:
seq_hash[( sequence_count, title )] = sequence
# return only those lengths are in the range
out = open( outfile, 'w' )
title_keys = seq_hash.keys()
title_keys.sort()
for i, fasta_title in title_keys:
sequence = seq_hash[( i, fasta_title )]
print >> out, "%s\t%s" %( fasta_title, sequence )
out.close()
if __name__ == "__main__" : __main__()
@@ -1,6 +1,6 @@
<tool id="fasta_tool_fasta2tab" name="FASTA to TAB">
<description>Converts a FASTA formated file to TAB format</description>
<command interpreter="python">fasta_tool_fasta2tab.py $input $output </command>
<tool id="fasta2tab" name="FASTA-to-Tabular" version="1.0.0">
<description>Converts a FASTA file to Tabular format</description>
<command interpreter="python">fasta_to_tabular.py $input $output</command>
<inputs>
<param name="input" type="data" format="fasta" label="Fasta file"/>
</inputs>
@@ -10,11 +10,11 @@
<tests>
<test>
<param name="input" value="454.fna" />
<output name="output" file="fasta_tool_convert2tab_1.out" />
<output name="output" file="fasta_to_tabular_out1.tabular" />
</test>
<test>
<param name="input" value="extract_genomic_dna.fa" />
<output name="output" file="fasta_tool_convert2tab_2.out" />
<param name="input" value="4.fasta" />
<output name="output" file="fasta_to_tabular_out2.tabular" />
</test>
</tests>
<help>
@@ -30,7 +30,7 @@ A fasta file with two sequences::
GGTGACATCGCCCACCACGGTACTCACTGGCTGGCTCTGGTTCCCGGCGGCATCGGAGGC
CACCACGTTGAGGGTATTCCCCTCGGTTTGTGGCTCGGTGAGAACCACGTTGTAGTCGCC
ATTGGTC
Returns::
&gt;EYKX4VC01B65GS length=54 xy=0784_1754 region=1 run=R_2007_11_07_16_15_57_ CCGGTATCCGGGTGCCGTGATGAGCGCCACCGGAACGAATTCGACTATGCCGAA
-58
View File
@@ -1,58 +0,0 @@
#! /usr/bin/python
"""
Input: fasta, minimal length, maximal length
Output: fasta
Return sequences whose lengths are within the range.
Wen-Yu Chung
"""
import sys, os
seq_hash = {}
def parse_fasta_format(file_handle):
# detect whether it's score or seq files
# return a hash: key = title and value = seq
# use seq_hash #tmp_hash = {}
tmp_title = ''
tmp_seq = ''
tmp_seq_count = 0
for i, each_line in enumerate(file_handle):
each_line = each_line.strip('\r\n')
if (each_line[0] == '>'):
if (len(tmp_seq) > 0):
tmp_seq_count += 1
seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq
tmp_title = each_line
tmp_seq = ''
else:
tmp_seq = tmp_seq + each_line
if (each_line.split()[0].isdigit()):
tmp_seq = tmp_seq + ' '
if (len(tmp_seq) > 0):
seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq
return 0
def __main__():
input_filename = sys.argv[1]
output_filename = sys.argv[2]
# parse fasta format into a hast table
input_handle = open(input_filename, 'r')
parse_fasta_format(input_handle)
input_handle.close()
# return only those lengths are in the range
output_handle = open(output_filename, 'w')
title_keys = seq_hash.keys()
title_keys.sort()
for (i, fasta_title) in title_keys:
tmp_seq = seq_hash[(i, fasta_title)]
print >> output_handle, "%s\t%s" %(fasta_title, tmp_seq)
output_handle.close()
return 0
if __name__ == "__main__" : __main__()
-59
View File
@@ -1,59 +0,0 @@
#! /usr/bin/python
"""
Input: fasta, minimal length, maximal length
Output: fasta
Return sequences whose lengths are within the range.
Wen-Yu Chung
"""
import sys, os
seq_hash = {}
def __main__():
input_filename = sys.argv[1]
if (sys.argv[2] != ""): input_titleCol = sys.argv[2]
else: print >> sys.stderr, "Please indicate columns for the titles and sequences."
input_seqCol = int(sys.argv[3]) - 1
output_filename = sys.argv[4]
if (input_seqCol < 0): print >> sys.stderr, "Column index starts from 1."
titleCols = input_titleCol.split(',')
# print input_titleCol
# print input_seqCol
input_handle = open(input_filename, 'r')
output_handle = open(output_filename, 'w')
# need to check which column is the title, which is the sequence
for i, line in enumerate(input_handle):
line = line.strip('\r\n')
fields = line.split()
total_columns = len(fields)
if (input_seqCol > total_columns): print >> sys.stderr, "Sequence column does not exist."
fasta_title = []
for j in titleCols:
if (j.isdigit()):
correct_index = int(j) - 1
if (correct_index < total_columns) and (correct_index >= 0):
fasta_title.append(fields[correct_index])
else:
print >> sys.stderr, "Title column does not exist. Column index starts from 1, not 0."
else:
print >> sys.stderr, "Title column is not an index."
fasta_seq = fields[input_seqCol]
if (fasta_title[0].startswith(">")): fasta_title[0]=fasta_title[0][1:]
print >> output_handle, ">%s\n%s" %("_".join(fasta_title), fasta_seq)
output_handle.close()
input_handle.close()
return 0
if __name__ == "__main__" : __main__()
+46
View File
@@ -0,0 +1,46 @@
#! /usr/bin/python
"""
Input: fasta, minimal length, maximal length
Output: fasta
Return sequences whose lengths are within the range.
Wen-Yu Chung
"""
import sys, os
seq_hash = {}
def stop_err( msg ):
sys.stderr.write( msg )
sys.exit()
def __main__():
infile = sys.argv[1]
title_col = sys.argv[2]
seq_col = sys.argv[3]
outfile = sys.argv[4]
if title_col == None or title_col == 'None' or seq_col == None or seq_col == 'None':
stop_err( "Columns not specified." )
try:
seq_col = int( seq_col ) - 1
except:
stop_err( "Invalid Sequence Column: %s." %seq_col )
title_col_list = title_col.split( ',' )
out = open( outfile, 'w' )
for i, line in enumerate( open( infile ) ):
line = line.strip( '\r\n' )
fields = line.split()
fasta_title = []
for j in title_col_list:
j = int( j ) - 1
fasta_title.append( fields[j] )
fasta_seq = fields[seq_col]
if ( fasta_title[0].startswith(">") ):
fasta_title[0] = fasta_title[0][1:]
print >> out, ">%s\n%s" % ( "_".join( fasta_title ), fasta_seq )
out.close()
if __name__ == "__main__" : __main__()
@@ -1,10 +1,10 @@
<tool id="fasta_tool_tab2fasta" name="TAB to FASTA">
<description>Converts a TAB formated file to FASTA format</description>
<command interpreter="python">fasta_tool_tab2fasta.py $input $input_titleCol $input_seqCol $output </command>
<tool id="tab2fasta" name="Tabular-to-FASTA" version="1.0.0">
<description>Converts a tabular file to FASTA format</description>
<command interpreter="python">tabular_to_fasta.py $input $title_col $seq_col $output </command>
<inputs>
<param name="input" type="data" format="tabular" label="Tab-delimited file"/>
<param name="input_titleCol" type="text" size="15" value="1" label="Which column(s) is the title" help="may use more than one column seperated by comma, ex. 1,2. Values will be concatenated by underline."/>
<param name="input_seqCol" type="integer" size="15" value="2" label="Which column is the sequence" />
<param name="title_col" type="data_column" data_ref="input" multiple="True" numerical="False" label="Title column(s)" help="Multi-select list - hold the appropriate key while clicking to select multiple columns"/>
<param name="seq_col" type="data_column" data_ref="input" numerical="False" label="Sequence column" />
</inputs>
<outputs>
<data name="output" format="fasta"/>
@@ -12,9 +12,9 @@
<tests>
<test>
<param name="input" value="solexa.fna" />
<param name="input_titleCol" value="1,2,3,4" />
<param name="input_seqCol" value="5" />
<output name="output" file="fasta_tool_tab2fasta_1.out" />
<param name="title_col" value="1,2,3,4" />
<param name="seq_col" value="5" />
<output name="output" file="tabular_to_fasta_out1.fasta" />
</test>
</tests>
<help>
@@ -22,15 +22,14 @@
**Example**
Solexa data::
5 300 902 419 GACTCATGATTTCTTACCTATTAGTGGTTGAACATC
5 300 880 431 GTGATATGTATGTTGACGGCCATAAGGCTGCTTCTT
5 300 896 461 GTTGTCGATAGAACTTCATGTGCCTGTAAAACAAGT
5 300 890 751 ACCAACCAGAACGTGAAAAAGCGTCCTGCGTGTAGC
5 300 897 443 GTTTATGTTGGTTTCATGGTTTTGTCTAACTTTATC
5 300 906 879 GCTTTACCGTCTTTCCAGAAATTGTTCCAAGTATCG
Choose **columns 3 and 4** (3,4) to be the fasta title, the tool returns::
Selecting **c3 and c4** as the Title Columns and **c5** as the Sequence Column will result in::
&gt;902_419
GACTCATGATTTCTTACCTATTAGTGGTTGAACATC