mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Fixes for tabular_to_fasta and fasta_to_tabular tools - requires update to datatype_converters_conf.xml
- Use ColumnListParameter rather than TextToolParameter - Use better naming convertions for code and test-data files - Added fasta_to_tabular converter to datatype_converters_conf.xml - Fixed tool-conf so the converters are displayed in the Converters tool section - Other miscellaneous cleanup.
This commit is contained in:
@@ -1,8 +1,9 @@
|
||||
<?xml version="1.0"?>
|
||||
<converters>
|
||||
<converter file="interval_to_bed_converter.xml" source_datatype="interval" target_datatype="bed"/>
|
||||
<converter file="maf_to_interval_converter.xml" source_datatype="maf" target_datatype="interval"/>
|
||||
<converter file="bed_to_gff_converter.xml" source_datatype="bed" target_datatype="gff"/>
|
||||
<converter file="fasta_to_tabular_converter.xml" source_datatype="fasta" target_datatype="tabular"/>
|
||||
<converter file="gff_to_bed_converter.xml" source_datatype="gff" target_datatype="bed"/>
|
||||
<converter file="interval_to_bed_converter.xml" source_datatype="interval" target_datatype="bed"/>
|
||||
<converter file="maf_to_fasta_converter.xml" source_datatype="maf" target_datatype="fasta"/>
|
||||
<converter file="maf_to_interval_converter.xml" source_datatype="maf" target_datatype="interval"/>
|
||||
</converters>
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
#! /usr/bin/python
|
||||
# This code exists in 2 places: ~/datatypes/converters and ~/tools/fasta_tools
|
||||
"""
|
||||
Input: fasta, minimal length, maximal length
|
||||
Output: fasta
|
||||
Return sequences whose lengths are within the range.
|
||||
Wen-Yu Chung
|
||||
"""
|
||||
|
||||
import sys, os
|
||||
|
||||
seq_hash = {}
|
||||
|
||||
def __main__():
|
||||
infile = sys.argv[1]
|
||||
outfile = sys.argv[2]
|
||||
title = ''
|
||||
sequence = ''
|
||||
sequence_count = 0
|
||||
for i, line in enumerate( open( infile ) ):
|
||||
line = line.rstrip( '\r\n' )
|
||||
if line.startswith( '>' ):
|
||||
if sequence:
|
||||
sequence_count += 1
|
||||
seq_hash[( sequence_count, title )] = sequence
|
||||
title = line
|
||||
sequence = ''
|
||||
else:
|
||||
sequence += line
|
||||
if line.split()[0].isdigit():
|
||||
sequence += ' '
|
||||
if sequence:
|
||||
seq_hash[( sequence_count, title )] = sequence
|
||||
# return only those lengths are in the range
|
||||
out = open( outfile, 'w' )
|
||||
title_keys = seq_hash.keys()
|
||||
title_keys.sort()
|
||||
for i, fasta_title in title_keys:
|
||||
sequence = seq_hash[( i, fasta_title )]
|
||||
print >> out, "%s\t%s" %( fasta_title, sequence )
|
||||
out.close()
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
@@ -0,0 +1,13 @@
|
||||
<tool id="CONVERTER_fasta_to_tabular" name="Convert FASTA to Tabular" version="1.0.0">
|
||||
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
|
||||
<!-- Used on the metadata edit page. -->
|
||||
<command interpreter="python">fasta_to_tabular_converter.py $input $output</command>
|
||||
<inputs>
|
||||
<param name="input" type="data" format="fasta" label="Fasta file"/>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data name="output" format="tabular"/>
|
||||
</outputs>
|
||||
<help>
|
||||
</help>
|
||||
</tool>
|
||||
@@ -69,8 +69,11 @@
|
||||
<tool file="patmat/findcluster_mysql.xml" />
|
||||
</section> -->
|
||||
<section name="Fetch Sequences" id="fetchSeq">
|
||||
<!--
|
||||
<tool file="extract/fasta-subseq-wrapper.xml" />
|
||||
<tool file="extract/twoBitToFa_wrapper.xml" />
|
||||
-->
|
||||
<tool file="extract/extract_genomic_dna.xml" />
|
||||
</section>
|
||||
<section name="Fetch Alignments" id="fetchAlign">
|
||||
<tool file="maf/interval2maf_pairwise.xml" />
|
||||
|
||||
@@ -55,14 +55,16 @@
|
||||
<tool file="stats/grouping.xml" />
|
||||
</section>
|
||||
<section name="Convert Formats" id="convert">
|
||||
<tool file="maf/maf_to_fasta.xml" />
|
||||
<tool file="maf/maf_to_bed.xml" />
|
||||
<tool file="filters/gff2bed.xml" />
|
||||
<tool file="filters/bed2gff.xml" />
|
||||
<tool file="filters/axt_to_fasta.xml" />
|
||||
<tool file="filters/axt_to_concat_fasta.xml" />
|
||||
<tool file="filters/axt_to_fasta.xml" />
|
||||
<tool file="filters/axt_to_lav.xml" />
|
||||
<tool file="filters/bed2gff.xml" />
|
||||
<tool file="fasta_tools/fasta_to_tabular.xml" />
|
||||
<tool file="filters/gff2bed.xml" />
|
||||
<tool file="filters/lav_to_bed.xml" />
|
||||
<tool file="maf/maf_to_bed.xml" />
|
||||
<tool file="maf/maf_to_fasta.xml" />
|
||||
<tool file="fasta_tools/tabular_to_fasta.xml" />
|
||||
</section>
|
||||
<section name="Extract Features" id="features">
|
||||
<tool file="filters/ucsc_gene_bed_to_exon_bed.xml" />
|
||||
@@ -303,8 +305,6 @@
|
||||
<section name="Fasta Tools" id="fasta_tools">
|
||||
<tool file="fasta_tools/fasta_tool_compute_length.xml" />
|
||||
<tool file="fasta_tools/fasta_tool_filter_by_length.xml" />
|
||||
<tool file="fasta_tools/fasta_tool_fasta2tab.xml" />
|
||||
<tool file="fasta_tools/fasta_tool_tab2fasta.xml" />
|
||||
</section>
|
||||
<section name="Metagenomics" id="metagenomics">
|
||||
<tool file="metag_tools/short_reads_trim_seq.xml" />
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
#! /usr/bin/python
|
||||
# This code exists in 2 places: ~/datatypes/converters and ~/tools/fasta_tools
|
||||
"""
|
||||
Input: fasta, minimal length, maximal length
|
||||
Output: fasta
|
||||
Return sequences whose lengths are within the range.
|
||||
Wen-Yu Chung
|
||||
"""
|
||||
|
||||
import sys, os
|
||||
|
||||
seq_hash = {}
|
||||
|
||||
def __main__():
|
||||
infile = sys.argv[1]
|
||||
outfile = sys.argv[2]
|
||||
title = ''
|
||||
sequence = ''
|
||||
sequence_count = 0
|
||||
for i, line in enumerate( open( infile ) ):
|
||||
line = line.rstrip( '\r\n' )
|
||||
if line.startswith( '>' ):
|
||||
if sequence:
|
||||
sequence_count += 1
|
||||
seq_hash[( sequence_count, title )] = sequence
|
||||
title = line
|
||||
sequence = ''
|
||||
else:
|
||||
sequence += line
|
||||
if line.split()[0].isdigit():
|
||||
sequence += ' '
|
||||
if sequence:
|
||||
seq_hash[( sequence_count, title )] = sequence
|
||||
# return only those lengths are in the range
|
||||
out = open( outfile, 'w' )
|
||||
title_keys = seq_hash.keys()
|
||||
title_keys.sort()
|
||||
for i, fasta_title in title_keys:
|
||||
sequence = seq_hash[( i, fasta_title )]
|
||||
print >> out, "%s\t%s" %( fasta_title, sequence )
|
||||
out.close()
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
@@ -1,6 +1,6 @@
|
||||
<tool id="fasta_tool_fasta2tab" name="FASTA to TAB">
|
||||
<description>Converts a FASTA formated file to TAB format</description>
|
||||
<command interpreter="python">fasta_tool_fasta2tab.py $input $output </command>
|
||||
<tool id="fasta2tab" name="FASTA-to-Tabular" version="1.0.0">
|
||||
<description>Converts a FASTA file to Tabular format</description>
|
||||
<command interpreter="python">fasta_to_tabular.py $input $output</command>
|
||||
<inputs>
|
||||
<param name="input" type="data" format="fasta" label="Fasta file"/>
|
||||
</inputs>
|
||||
@@ -10,11 +10,11 @@
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input" value="454.fna" />
|
||||
<output name="output" file="fasta_tool_convert2tab_1.out" />
|
||||
<output name="output" file="fasta_to_tabular_out1.tabular" />
|
||||
</test>
|
||||
<test>
|
||||
<param name="input" value="extract_genomic_dna.fa" />
|
||||
<output name="output" file="fasta_tool_convert2tab_2.out" />
|
||||
<param name="input" value="4.fasta" />
|
||||
<output name="output" file="fasta_to_tabular_out2.tabular" />
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
@@ -30,7 +30,7 @@ A fasta file with two sequences::
|
||||
GGTGACATCGCCCACCACGGTACTCACTGGCTGGCTCTGGTTCCCGGCGGCATCGGAGGC
|
||||
CACCACGTTGAGGGTATTCCCCTCGGTTTGTGGCTCGGTGAGAACCACGTTGTAGTCGCC
|
||||
ATTGGTC
|
||||
|
||||
|
||||
Returns::
|
||||
|
||||
>EYKX4VC01B65GS length=54 xy=0784_1754 region=1 run=R_2007_11_07_16_15_57_ CCGGTATCCGGGTGCCGTGATGAGCGCCACCGGAACGAATTCGACTATGCCGAA
|
||||
@@ -1,58 +0,0 @@
|
||||
#! /usr/bin/python
|
||||
"""
|
||||
Input: fasta, minimal length, maximal length
|
||||
Output: fasta
|
||||
Return sequences whose lengths are within the range.
|
||||
Wen-Yu Chung
|
||||
"""
|
||||
|
||||
import sys, os
|
||||
|
||||
seq_hash = {}
|
||||
|
||||
def parse_fasta_format(file_handle):
|
||||
# detect whether it's score or seq files
|
||||
# return a hash: key = title and value = seq
|
||||
|
||||
# use seq_hash #tmp_hash = {}
|
||||
tmp_title = ''
|
||||
tmp_seq = ''
|
||||
tmp_seq_count = 0
|
||||
for i, each_line in enumerate(file_handle):
|
||||
each_line = each_line.strip('\r\n')
|
||||
if (each_line[0] == '>'):
|
||||
if (len(tmp_seq) > 0):
|
||||
tmp_seq_count += 1
|
||||
seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq
|
||||
tmp_title = each_line
|
||||
tmp_seq = ''
|
||||
else:
|
||||
tmp_seq = tmp_seq + each_line
|
||||
if (each_line.split()[0].isdigit()):
|
||||
tmp_seq = tmp_seq + ' '
|
||||
if (len(tmp_seq) > 0):
|
||||
seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq
|
||||
|
||||
return 0
|
||||
|
||||
def __main__():
|
||||
|
||||
input_filename = sys.argv[1]
|
||||
output_filename = sys.argv[2]
|
||||
|
||||
# parse fasta format into a hast table
|
||||
input_handle = open(input_filename, 'r')
|
||||
parse_fasta_format(input_handle)
|
||||
input_handle.close()
|
||||
|
||||
# return only those lengths are in the range
|
||||
output_handle = open(output_filename, 'w')
|
||||
title_keys = seq_hash.keys()
|
||||
title_keys.sort()
|
||||
for (i, fasta_title) in title_keys:
|
||||
tmp_seq = seq_hash[(i, fasta_title)]
|
||||
print >> output_handle, "%s\t%s" %(fasta_title, tmp_seq)
|
||||
output_handle.close()
|
||||
return 0
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
@@ -1,59 +0,0 @@
|
||||
#! /usr/bin/python
|
||||
"""
|
||||
Input: fasta, minimal length, maximal length
|
||||
Output: fasta
|
||||
Return sequences whose lengths are within the range.
|
||||
Wen-Yu Chung
|
||||
"""
|
||||
|
||||
import sys, os
|
||||
|
||||
seq_hash = {}
|
||||
|
||||
def __main__():
|
||||
|
||||
input_filename = sys.argv[1]
|
||||
if (sys.argv[2] != ""): input_titleCol = sys.argv[2]
|
||||
else: print >> sys.stderr, "Please indicate columns for the titles and sequences."
|
||||
input_seqCol = int(sys.argv[3]) - 1
|
||||
output_filename = sys.argv[4]
|
||||
|
||||
if (input_seqCol < 0): print >> sys.stderr, "Column index starts from 1."
|
||||
|
||||
titleCols = input_titleCol.split(',')
|
||||
|
||||
# print input_titleCol
|
||||
# print input_seqCol
|
||||
|
||||
input_handle = open(input_filename, 'r')
|
||||
output_handle = open(output_filename, 'w')
|
||||
|
||||
# need to check which column is the title, which is the sequence
|
||||
for i, line in enumerate(input_handle):
|
||||
line = line.strip('\r\n')
|
||||
fields = line.split()
|
||||
total_columns = len(fields)
|
||||
|
||||
if (input_seqCol > total_columns): print >> sys.stderr, "Sequence column does not exist."
|
||||
|
||||
fasta_title = []
|
||||
for j in titleCols:
|
||||
if (j.isdigit()):
|
||||
correct_index = int(j) - 1
|
||||
if (correct_index < total_columns) and (correct_index >= 0):
|
||||
fasta_title.append(fields[correct_index])
|
||||
else:
|
||||
print >> sys.stderr, "Title column does not exist. Column index starts from 1, not 0."
|
||||
else:
|
||||
print >> sys.stderr, "Title column is not an index."
|
||||
|
||||
fasta_seq = fields[input_seqCol]
|
||||
if (fasta_title[0].startswith(">")): fasta_title[0]=fasta_title[0][1:]
|
||||
print >> output_handle, ">%s\n%s" %("_".join(fasta_title), fasta_seq)
|
||||
|
||||
output_handle.close()
|
||||
input_handle.close()
|
||||
|
||||
return 0
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
@@ -0,0 +1,46 @@
|
||||
#! /usr/bin/python
|
||||
"""
|
||||
Input: fasta, minimal length, maximal length
|
||||
Output: fasta
|
||||
Return sequences whose lengths are within the range.
|
||||
Wen-Yu Chung
|
||||
"""
|
||||
|
||||
import sys, os
|
||||
|
||||
seq_hash = {}
|
||||
|
||||
def stop_err( msg ):
|
||||
sys.stderr.write( msg )
|
||||
sys.exit()
|
||||
|
||||
def __main__():
|
||||
infile = sys.argv[1]
|
||||
title_col = sys.argv[2]
|
||||
seq_col = sys.argv[3]
|
||||
outfile = sys.argv[4]
|
||||
|
||||
if title_col == None or title_col == 'None' or seq_col == None or seq_col == 'None':
|
||||
stop_err( "Columns not specified." )
|
||||
try:
|
||||
seq_col = int( seq_col ) - 1
|
||||
except:
|
||||
stop_err( "Invalid Sequence Column: %s." %seq_col )
|
||||
|
||||
title_col_list = title_col.split( ',' )
|
||||
out = open( outfile, 'w' )
|
||||
|
||||
for i, line in enumerate( open( infile ) ):
|
||||
line = line.strip( '\r\n' )
|
||||
fields = line.split()
|
||||
fasta_title = []
|
||||
for j in title_col_list:
|
||||
j = int( j ) - 1
|
||||
fasta_title.append( fields[j] )
|
||||
fasta_seq = fields[seq_col]
|
||||
if ( fasta_title[0].startswith(">") ):
|
||||
fasta_title[0] = fasta_title[0][1:]
|
||||
print >> out, ">%s\n%s" % ( "_".join( fasta_title ), fasta_seq )
|
||||
out.close()
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
+10
-11
@@ -1,10 +1,10 @@
|
||||
<tool id="fasta_tool_tab2fasta" name="TAB to FASTA">
|
||||
<description>Converts a TAB formated file to FASTA format</description>
|
||||
<command interpreter="python">fasta_tool_tab2fasta.py $input $input_titleCol $input_seqCol $output </command>
|
||||
<tool id="tab2fasta" name="Tabular-to-FASTA" version="1.0.0">
|
||||
<description>Converts a tabular file to FASTA format</description>
|
||||
<command interpreter="python">tabular_to_fasta.py $input $title_col $seq_col $output </command>
|
||||
<inputs>
|
||||
<param name="input" type="data" format="tabular" label="Tab-delimited file"/>
|
||||
<param name="input_titleCol" type="text" size="15" value="1" label="Which column(s) is the title" help="may use more than one column seperated by comma, ex. 1,2. Values will be concatenated by underline."/>
|
||||
<param name="input_seqCol" type="integer" size="15" value="2" label="Which column is the sequence" />
|
||||
<param name="title_col" type="data_column" data_ref="input" multiple="True" numerical="False" label="Title column(s)" help="Multi-select list - hold the appropriate key while clicking to select multiple columns"/>
|
||||
<param name="seq_col" type="data_column" data_ref="input" numerical="False" label="Sequence column" />
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data name="output" format="fasta"/>
|
||||
@@ -12,9 +12,9 @@
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input" value="solexa.fna" />
|
||||
<param name="input_titleCol" value="1,2,3,4" />
|
||||
<param name="input_seqCol" value="5" />
|
||||
<output name="output" file="fasta_tool_tab2fasta_1.out" />
|
||||
<param name="title_col" value="1,2,3,4" />
|
||||
<param name="seq_col" value="5" />
|
||||
<output name="output" file="tabular_to_fasta_out1.fasta" />
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
@@ -22,15 +22,14 @@
|
||||
**Example**
|
||||
|
||||
Solexa data::
|
||||
|
||||
5 300 902 419 GACTCATGATTTCTTACCTATTAGTGGTTGAACATC
|
||||
5 300 880 431 GTGATATGTATGTTGACGGCCATAAGGCTGCTTCTT
|
||||
5 300 896 461 GTTGTCGATAGAACTTCATGTGCCTGTAAAACAAGT
|
||||
5 300 890 751 ACCAACCAGAACGTGAAAAAGCGTCCTGCGTGTAGC
|
||||
5 300 897 443 GTTTATGTTGGTTTCATGGTTTTGTCTAACTTTATC
|
||||
5 300 906 879 GCTTTACCGTCTTTCCAGAAATTGTTCCAAGTATCG
|
||||
|
||||
Choose **columns 3 and 4** (3,4) to be the fasta title, the tool returns::
|
||||
|
||||
Selecting **c3 and c4** as the Title Columns and **c5** as the Sequence Column will result in::
|
||||
|
||||
>902_419
|
||||
GACTCATGATTTCTTACCTATTAGTGGTTGAACATC
|
||||
Reference in New Issue
Block a user