mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
More clean up and modification of FASTA tools
This commit is contained in:
@@ -56,6 +56,13 @@
|
||||
<tool file="maf/maf_to_fasta.xml" />
|
||||
<tool file="fasta_tools/tabular_to_fasta.xml" />
|
||||
</section>
|
||||
<section name="FASTA manipulation" id="fasta_manipulation">
|
||||
<tool file="fasta_tools/fasta_compute_length.xml" />
|
||||
<tool file="fasta_tools/fasta_filter_by_length.xml" />
|
||||
<tool file="fasta_tools/fasta_concatenate_by_species.xml" />
|
||||
<tool file="fasta_tools/fasta_to_tabular.xml" />
|
||||
<tool file="fasta_tools/tabular_to_fasta.xml" />
|
||||
</section>
|
||||
<section name="Extract Features" id="features">
|
||||
<tool file="filters/ucsc_gene_bed_to_exon_bed.xml" />
|
||||
<tool file="extract/extract_GFF_Features.xml" />
|
||||
|
||||
@@ -264,6 +264,8 @@
|
||||
<tool file="fasta_tools/fasta_compute_length.xml" />
|
||||
<tool file="fasta_tools/fasta_filter_by_length.xml" />
|
||||
<tool file="fasta_tools/fasta_concatenate_by_species.xml" />
|
||||
<tool file="fasta_tools/fasta_to_tabular.xml" />
|
||||
<tool file="fasta_tools/tabular_to_fasta.xml" />
|
||||
</section>
|
||||
<section name="Short Read Analysis" id="short_read_analysis">
|
||||
<tool file="metag_tools/short_reads_figure_score.xml" />
|
||||
|
||||
@@ -12,10 +12,14 @@ assert sys.version_info[:2] >= ( 2, 4 )
|
||||
def __main__():
|
||||
input_filename = sys.argv[1]
|
||||
output_filename = sys.argv[2]
|
||||
keep_first = int( sys.argv[3] ) + 1
|
||||
tmp_title = tmp_seq = ''
|
||||
tmp_seq_count = 0
|
||||
seq_hash = {}
|
||||
|
||||
if keep_first == 0:
|
||||
keep_first = None
|
||||
|
||||
for i, line in enumerate( file( input_filename ) ):
|
||||
line = line.rstrip( '\r\n' )
|
||||
if not line or line.startswith( '#' ):
|
||||
@@ -38,7 +42,7 @@ def __main__():
|
||||
output_handle = open( output_filename, 'w' )
|
||||
for i, fasta_title in title_keys:
|
||||
tmp_seq = seq_hash[ ( i, fasta_title ) ]
|
||||
output_handle.write( "%s\t%d\n" % ( fasta_title[ 1: ], len( tmp_seq ) ) )
|
||||
output_handle.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( tmp_seq ) ) )
|
||||
output_handle.close()
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
@@ -1,8 +1,9 @@
|
||||
<tool id="fasta_compute_length" name="Count FASTA Length">
|
||||
<description> </description>
|
||||
<command interpreter="python">fasta_compute_length.py $input $output </command>
|
||||
<tool id="fasta_compute_length" name="Compute">
|
||||
<description>sequence length</description>
|
||||
<command interpreter="python">fasta_compute_length.py $input $output $keep_first</command>
|
||||
<inputs>
|
||||
<param name="input" type="data" format="fasta" label="Fasta file"/>
|
||||
<param name="input" type="data" format="fasta" label="Compute length for these sequences"/>
|
||||
<param name="keep_first" type="integer" size="5" value="0" label="How many title characters to keep?" help="'0' = keep the whole thing"/>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data name="output" format="tabular"/>
|
||||
@@ -10,10 +11,12 @@
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input" value="454.fasta" />
|
||||
<param name="keep_first" value="0"/>
|
||||
<output name="output" file="fasta_tool_compute_length_1.out" />
|
||||
</test>
|
||||
<test>
|
||||
<param name="input" value="extract_genomic_dna_out1.fasta" />
|
||||
<param name="keep_first" value="0"/>
|
||||
<output name="output" file="fasta_tool_compute_length_2.out" />
|
||||
</test>
|
||||
</tests>
|
||||
@@ -21,27 +24,21 @@
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool counts the length of each fasta sequence in the file. The output file has two columns per line (separated by tab): fasta titles and lengths of the sequences.
|
||||
This tool counts the length of each fasta sequence in the file. The output file has two columns per line (separated by tab): fasta titles and lengths of the sequences. The option *How many characters to keep?* allows to select a specified number of letters from the beginning of each FASTA entry.
|
||||
|
||||
-----
|
||||
|
||||
**Example**
|
||||
|
||||
- assume the input file contains fasta sequences::
|
||||
Suppose you have the following FASTA formatted sequences from a Roche (454) FLX sequencing run::
|
||||
|
||||
>seq1
|
||||
TCATTTA
|
||||
>seq2
|
||||
ATGGCGTCGGCC
|
||||
>seq3
|
||||
TCACATGATG
|
||||
>EYKX4VC02EQLO5 length=108 xy=1826_0455 region=2 run=R_2007_11_07_16_15_57_␍ TCCGCGCCGAGCATGCCCATCTTGGATTCCGGCGCGATGACCATCGCCCGCTCCACCACG␍ TTCGGCCGGCCCTTCTCGTCGAGGAATGACACCAGCGCTTCGCCCACG␍ >EYKX4VC02D4GS2 length=60 xy=1573_3972 region=2 run=R_2007_11_07_16_15_57_␍ AATAAAACTAAATCAGCAAAGACTGGCAAATACTCACAGGCTTATACAATACAAATGTAA
|
||||
|
||||
- the tool will return(first column is the titles, second column is the length of the sequences)::
|
||||
|
||||
>seq1 7
|
||||
>seq2 12
|
||||
>seq3 10
|
||||
Running this tool while setting **How many characters to keep?** to **14** will produce this::
|
||||
|
||||
EYKX4VC02EQLO5 108
|
||||
EYKX4VC02D4GS2 60
|
||||
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
@@ -1,5 +1,5 @@
|
||||
<tool id="fasta_concatenate0" name="Concatenate FASTA" version="0.0.0">
|
||||
<description>alignment by species</description>
|
||||
<tool id="fasta_concatenate0" name="Concatenate" version="0.0.0">
|
||||
<description>FASTA alignment by species</description>
|
||||
<command interpreter="python">fasta_concatenate_by_species.py $input1 $out_file1</command>
|
||||
<inputs>
|
||||
<param name="input1" type="data" format="fasta" label="FASTA alignment"/>
|
||||
@@ -14,9 +14,14 @@
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
This tools attempts to parse FASTA headers to determine the species for each sequence in a multiple FASTA alignment.
|
||||
It then linearly concatenates the sequences for each species in the file, creating one sequence per determined species.
|
||||
|
||||
-------
|
||||
|
||||
**Example**
|
||||
|
||||
Starting FASTA::
|
||||
@@ -45,7 +50,7 @@ Starting FASTA::
|
||||
|
||||
|
||||
|
||||
Becomes::
|
||||
becomes::
|
||||
|
||||
>hg18
|
||||
GTGT
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
<tool id="fasta_filter_by_length" name="Filter FASTA by Length">
|
||||
<description> </description>
|
||||
<tool id="fasta_filter_by_length" name="Filter">
|
||||
<description>sequences by length</description>
|
||||
<command interpreter="python">fasta_filter_by_length.py $input $min_length $max_length $output </command>
|
||||
<inputs>
|
||||
<param name="input" type="data" format="fasta" label="Fasta file"/>
|
||||
<param name="min_length" type="integer" size="15" value="0" label="Minimal length of the return sequence" />
|
||||
<param name="max_length" type="integer" size="15" value="0" label="Maximum length of the return sequence" help="no limitation if 0"/>
|
||||
<param name="min_length" type="integer" size="15" value="0" label="Minimal length" />
|
||||
<param name="max_length" type="integer" size="15" value="0" label="Maximum length" help="Setting to '0' will return all sequences longer than the 'Minimal length'"/>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data name="output" format="fasta"/>
|
||||
@@ -21,19 +21,19 @@
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
**TIP**. If only want to show sequences longer than a threshold, set *minimal length* to the threshold and leave *maximum length* to zero.
|
||||
**TIP**. To return sequences longer than a certain length, set *Minimal length* to desired value and leave *Maximum length* set to '0'.
|
||||
|
||||
-----
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool accepts two parameters: *minimal length* and *maximum length*, and returns sequences of length within the two thresholds.
|
||||
Outputs sequences between *Minimal length* and *Maximum length*.
|
||||
|
||||
-----
|
||||
|
||||
**Example**
|
||||
|
||||
- assume the input file contains fasta sequences::
|
||||
Suppose you have the following FASTA formatted sequences::
|
||||
|
||||
>seq1
|
||||
TCATTTAATGAC
|
||||
@@ -44,7 +44,7 @@
|
||||
>seq4
|
||||
ATGGAAGC
|
||||
|
||||
- return sequences with length longer than 10bp (set the *minimal length* to 10, and the *maximum length* to 0 (no limitation))::
|
||||
Setting the **Minimal length** to **10**, and the **Maximum length** to **0** will return all sequences longer than 10 bp::
|
||||
|
||||
>seq1
|
||||
TCATTTAATGAC
|
||||
|
||||
@@ -13,9 +13,14 @@ seq_hash = {}
|
||||
def __main__():
|
||||
infile = sys.argv[1]
|
||||
outfile = sys.argv[2]
|
||||
keep_first = int( sys.argv[3] ) + 1
|
||||
title = ''
|
||||
sequence = ''
|
||||
sequence_count = 0
|
||||
|
||||
if keep_first == 0:
|
||||
keep_first = None
|
||||
|
||||
for i, line in enumerate( open( infile ) ):
|
||||
line = line.rstrip( '\r\n' )
|
||||
if not line or line.startswith( '#' ):
|
||||
@@ -38,7 +43,7 @@ def __main__():
|
||||
out = open( outfile, 'w' )
|
||||
for i, fasta_title in title_keys:
|
||||
sequence = seq_hash[( i, fasta_title )]
|
||||
out.write( "%s\t%s\n" %( fasta_title, sequence ) )
|
||||
out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], sequence ) )
|
||||
out.close()
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
@@ -1,8 +1,9 @@
|
||||
<tool id="fasta2tab" name="FASTA-to-Tabular" version="1.0.0">
|
||||
<description>Converts a FASTA file to Tabular format</description>
|
||||
<command interpreter="python">fasta_to_tabular.py $input $output</command>
|
||||
<description>converts FASTA file to tabular format</description>
|
||||
<command interpreter="python">fasta_to_tabular.py $input $output $keep_first</command>
|
||||
<inputs>
|
||||
<param name="input" type="data" format="fasta" label="Fasta file"/>
|
||||
<param name="input" type="data" format="fasta" label="Convert these sequences"/>
|
||||
<param name="keep_first" type="integer" size="5" value="0" label="How many title characters to keep?" help="'0' = keep the whole thing"/>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data name="output" format="tabular"/>
|
||||
@@ -10,28 +11,34 @@
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input" value="454.fasta" />
|
||||
<param name="keep_first" value="0"/>
|
||||
<output name="output" file="fasta_to_tabular_out1.tabular" />
|
||||
</test>
|
||||
<test>
|
||||
<param name="input" value="4.fasta" />
|
||||
<param name="keep_first" value="0"/>
|
||||
<output name="output" file="fasta_to_tabular_out2.tabular" />
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool converts FASTA formatted sequences to TAB-delimited format. The option *How many characters to keep?* allows to select a specified number of letters from the beginning of each FASTA entry.
|
||||
|
||||
-----
|
||||
|
||||
**Example**
|
||||
|
||||
A fasta file with two sequences::
|
||||
Suppose you have the following FASTA formatted sequences from a Roche (454) FLX sequencing run::
|
||||
|
||||
>seq1
|
||||
CCGGTATCCG
|
||||
>seq2
|
||||
CTTACC
|
||||
>EYKX4VC02EQLO5 length=108 xy=1826_0455 region=2 run=R_2007_11_07_16_15_57_␍ TCCGCGCCGAGCATGCCCATCTTGGATTCCGGCGCGATGACCATCGCCCGCTCCACCACG␍ TTCGGCCGGCCCTTCTCGTCGAGGAATGACACCAGCGCTTCGCCCACG␍ >EYKX4VC02D4GS2 length=60 xy=1573_3972 region=2 run=R_2007_11_07_16_15_57_␍ AATAAAACTAAATCAGCAAAGACTGGCAAATACTCACAGGCTTATACAATACAAATGTAA
|
||||
|
||||
Returns::
|
||||
Running this tool while setting **How many characters to keep?** to **14** will produce this::
|
||||
|
||||
EYKX4VC02EQLO5 TCCGCGCCGAGCATGCCCATCTTGGATTCCGGCGCGATGACCATCGCCCGCTCCACCACGTTCGGCCGGCCCTTCTCGTCGAGGAATGACACCAGCGCTTCGCCCACG
|
||||
EYKX4VC02D4GS2 AATAAAACTAAATCAGCAAAGACTGGCAAATACTCACAGGCTTATACAATACAAATGTAA
|
||||
|
||||
>seq1 CCGGTATCCG
|
||||
>seq2 CTTACC
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
@@ -1,5 +1,5 @@
|
||||
<tool id="tab2fasta" name="Tabular-to-FASTA" version="1.1.0">
|
||||
<description>Converts a tabular file to FASTA format</description>
|
||||
<description>converts tabular file to FASTA format</description>
|
||||
<command interpreter="python">tabular_to_fasta.py $input $title_col $seq_col $output </command>
|
||||
<inputs>
|
||||
<param name="input" type="data" format="tabular" label="Tab-delimited file"/>
|
||||
@@ -19,14 +19,20 @@
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
Converts tab delimited data into FASTA formatted sequences.
|
||||
|
||||
-----------
|
||||
|
||||
**Example**
|
||||
|
||||
Solexa data::
|
||||
Suppose this is a sequence file produced by Illumina (Solexa) sequencer::
|
||||
|
||||
5 300 902 419 GACTCATGATTTCTTACCTATTAGTGGTTGAACATC
|
||||
5 300 880 431 GTGATATGTATGTTGACGGCCATAAGGCTGCTTCTT
|
||||
|
||||
Selecting **c3 and c4** as the Title Columns and **c5** as the Sequence Column will result in::
|
||||
Selecting **c3** and **c4** as the **Title column(s)** and **c5** as the **Sequence column** will result in::
|
||||
|
||||
>902_419
|
||||
GACTCATGATTTCTTACCTATTAGTGGTTGAACATC
|
||||
|
||||
Reference in New Issue
Block a user