diff --git a/tool_conf.xml.main b/tool_conf.xml.main
index ebbc28386fb..1064796a325 100644
--- a/tool_conf.xml.main
+++ b/tool_conf.xml.main
@@ -56,6 +56,13 @@
+
diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample
index d5849300b20..3db546f2bb6 100644
--- a/tool_conf.xml.sample
+++ b/tool_conf.xml.sample
@@ -264,6 +264,8 @@
+
+
diff --git a/tools/fasta_tools/fasta_compute_length.py b/tools/fasta_tools/fasta_compute_length.py
index 650deda0b5f..9bd36225a37 100644
--- a/tools/fasta_tools/fasta_compute_length.py
+++ b/tools/fasta_tools/fasta_compute_length.py
@@ -12,10 +12,14 @@ assert sys.version_info[:2] >= ( 2, 4 )
def __main__():
input_filename = sys.argv[1]
output_filename = sys.argv[2]
+ keep_first = int( sys.argv[3] ) + 1
tmp_title = tmp_seq = ''
tmp_seq_count = 0
seq_hash = {}
+ if keep_first == 0:
+ keep_first = None
+
for i, line in enumerate( file( input_filename ) ):
line = line.rstrip( '\r\n' )
if not line or line.startswith( '#' ):
@@ -38,7 +42,7 @@ def __main__():
output_handle = open( output_filename, 'w' )
for i, fasta_title in title_keys:
tmp_seq = seq_hash[ ( i, fasta_title ) ]
- output_handle.write( "%s\t%d\n" % ( fasta_title[ 1: ], len( tmp_seq ) ) )
+ output_handle.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( tmp_seq ) ) )
output_handle.close()
if __name__ == "__main__" : __main__()
\ No newline at end of file
diff --git a/tools/fasta_tools/fasta_compute_length.xml b/tools/fasta_tools/fasta_compute_length.xml
index 10e203f67c9..3778a7ccccd 100644
--- a/tools/fasta_tools/fasta_compute_length.xml
+++ b/tools/fasta_tools/fasta_compute_length.xml
@@ -1,8 +1,9 @@
-
-
- fasta_compute_length.py $input $output
+
+ sequence length
+ fasta_compute_length.py $input $output $keep_first
-
+
+
@@ -10,10 +11,12 @@
+
+
@@ -21,27 +24,21 @@
**What it does**
- This tool counts the length of each fasta sequence in the file. The output file has two columns per line (separated by tab): fasta titles and lengths of the sequences.
+This tool counts the length of each fasta sequence in the file. The output file has two columns per line (separated by tab): fasta titles and lengths of the sequences. The option *How many characters to keep?* allows to select a specified number of letters from the beginning of each FASTA entry.
-----
**Example**
-- assume the input file contains fasta sequences::
+Suppose you have the following FASTA formatted sequences from a Roche (454) FLX sequencing run::
- >seq1
- TCATTTA
- >seq2
- ATGGCGTCGGCC
- >seq3
- TCACATGATG
+ >EYKX4VC02EQLO5 length=108 xy=1826_0455 region=2 run=R_2007_11_07_16_15_57_
TCCGCGCCGAGCATGCCCATCTTGGATTCCGGCGCGATGACCATCGCCCGCTCCACCACG
TTCGGCCGGCCCTTCTCGTCGAGGAATGACACCAGCGCTTCGCCCACG
>EYKX4VC02D4GS2 length=60 xy=1573_3972 region=2 run=R_2007_11_07_16_15_57_
AATAAAACTAAATCAGCAAAGACTGGCAAATACTCACAGGCTTATACAATACAAATGTAA
-- the tool will return(first column is the titles, second column is the length of the sequences)::
-
- >seq1 7
- >seq2 12
- >seq3 10
+Running this tool while setting **How many characters to keep?** to **14** will produce this::
+ EYKX4VC02EQLO5 108
+ EYKX4VC02D4GS2 60
+
\ No newline at end of file
diff --git a/tools/fasta_tools/fasta_concatenate_by_species.xml b/tools/fasta_tools/fasta_concatenate_by_species.xml
index f78aab51d93..f85976c3574 100644
--- a/tools/fasta_tools/fasta_concatenate_by_species.xml
+++ b/tools/fasta_tools/fasta_concatenate_by_species.xml
@@ -1,5 +1,5 @@
-
- alignment by species
+
+ FASTA alignment by species
fasta_concatenate_by_species.py $input1 $out_file1
@@ -14,9 +14,14 @@
+
+**What it does**
+
This tools attempts to parse FASTA headers to determine the species for each sequence in a multiple FASTA alignment.
It then linearly concatenates the sequences for each species in the file, creating one sequence per determined species.
+-------
+
**Example**
Starting FASTA::
@@ -45,7 +50,7 @@ Starting FASTA::
-Becomes::
+becomes::
>hg18
GTGT
diff --git a/tools/fasta_tools/fasta_filter_by_length.xml b/tools/fasta_tools/fasta_filter_by_length.xml
index 27200919ab4..464d56db317 100644
--- a/tools/fasta_tools/fasta_filter_by_length.xml
+++ b/tools/fasta_tools/fasta_filter_by_length.xml
@@ -1,10 +1,10 @@
-
-
+
+ sequences by length
fasta_filter_by_length.py $input $min_length $max_length $output
-
-
+
+
@@ -21,19 +21,19 @@
.. class:: infomark
-**TIP**. If only want to show sequences longer than a threshold, set *minimal length* to the threshold and leave *maximum length* to zero.
+**TIP**. To return sequences longer than a certain length, set *Minimal length* to desired value and leave *Maximum length* set to '0'.
-----
**What it does**
- This tool accepts two parameters: *minimal length* and *maximum length*, and returns sequences of length within the two thresholds.
+Outputs sequences between *Minimal length* and *Maximum length*.
-----
**Example**
-- assume the input file contains fasta sequences::
+Suppose you have the following FASTA formatted sequences::
>seq1
TCATTTAATGAC
@@ -44,7 +44,7 @@
>seq4
ATGGAAGC
-- return sequences with length longer than 10bp (set the *minimal length* to 10, and the *maximum length* to 0 (no limitation))::
+Setting the **Minimal length** to **10**, and the **Maximum length** to **0** will return all sequences longer than 10 bp::
>seq1
TCATTTAATGAC
diff --git a/tools/fasta_tools/fasta_to_tabular.py b/tools/fasta_tools/fasta_to_tabular.py
index 26e538f6a71..c1e8eaede1f 100644
--- a/tools/fasta_tools/fasta_to_tabular.py
+++ b/tools/fasta_tools/fasta_to_tabular.py
@@ -13,9 +13,14 @@ seq_hash = {}
def __main__():
infile = sys.argv[1]
outfile = sys.argv[2]
+ keep_first = int( sys.argv[3] ) + 1
title = ''
sequence = ''
sequence_count = 0
+
+ if keep_first == 0:
+ keep_first = None
+
for i, line in enumerate( open( infile ) ):
line = line.rstrip( '\r\n' )
if not line or line.startswith( '#' ):
@@ -38,7 +43,7 @@ def __main__():
out = open( outfile, 'w' )
for i, fasta_title in title_keys:
sequence = seq_hash[( i, fasta_title )]
- out.write( "%s\t%s\n" %( fasta_title, sequence ) )
+ out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], sequence ) )
out.close()
if __name__ == "__main__" : __main__()
\ No newline at end of file
diff --git a/tools/fasta_tools/fasta_to_tabular.xml b/tools/fasta_tools/fasta_to_tabular.xml
index e0f5a742f70..4d8fdae6652 100644
--- a/tools/fasta_tools/fasta_to_tabular.xml
+++ b/tools/fasta_tools/fasta_to_tabular.xml
@@ -1,8 +1,9 @@
- Converts a FASTA file to Tabular format
- fasta_to_tabular.py $input $output
+ converts FASTA file to tabular format
+ fasta_to_tabular.py $input $output $keep_first
-
+
+
@@ -10,28 +11,34 @@
+
+
+**What it does**
+
+This tool converts FASTA formatted sequences to TAB-delimited format. The option *How many characters to keep?* allows to select a specified number of letters from the beginning of each FASTA entry.
+
+-----
+
**Example**
-A fasta file with two sequences::
+Suppose you have the following FASTA formatted sequences from a Roche (454) FLX sequencing run::
- >seq1
- CCGGTATCCG
- >seq2
- CTTACC
+ >EYKX4VC02EQLO5 length=108 xy=1826_0455 region=2 run=R_2007_11_07_16_15_57_
TCCGCGCCGAGCATGCCCATCTTGGATTCCGGCGCGATGACCATCGCCCGCTCCACCACG
TTCGGCCGGCCCTTCTCGTCGAGGAATGACACCAGCGCTTCGCCCACG
>EYKX4VC02D4GS2 length=60 xy=1573_3972 region=2 run=R_2007_11_07_16_15_57_
AATAAAACTAAATCAGCAAAGACTGGCAAATACTCACAGGCTTATACAATACAAATGTAA
-Returns::
+Running this tool while setting **How many characters to keep?** to **14** will produce this::
+
+ EYKX4VC02EQLO5 TCCGCGCCGAGCATGCCCATCTTGGATTCCGGCGCGATGACCATCGCCCGCTCCACCACGTTCGGCCGGCCCTTCTCGTCGAGGAATGACACCAGCGCTTCGCCCACG
+ EYKX4VC02D4GS2 AATAAAACTAAATCAGCAAAGACTGGCAAATACTCACAGGCTTATACAATACAAATGTAA
- >seq1 CCGGTATCCG
- >seq2 CTTACC
\ No newline at end of file
diff --git a/tools/fasta_tools/tabular_to_fasta.xml b/tools/fasta_tools/tabular_to_fasta.xml
index 1e4a0e87700..75faaf1e715 100644
--- a/tools/fasta_tools/tabular_to_fasta.xml
+++ b/tools/fasta_tools/tabular_to_fasta.xml
@@ -1,5 +1,5 @@
- Converts a tabular file to FASTA format
+ converts tabular file to FASTA format
tabular_to_fasta.py $input $title_col $seq_col $output
@@ -19,14 +19,20 @@
+**What it does**
+
+Converts tab delimited data into FASTA formatted sequences.
+
+-----------
+
**Example**
-Solexa data::
+Suppose this is a sequence file produced by Illumina (Solexa) sequencer::
5 300 902 419 GACTCATGATTTCTTACCTATTAGTGGTTGAACATC
5 300 880 431 GTGATATGTATGTTGACGGCCATAAGGCTGCTTCTT
-Selecting **c3 and c4** as the Title Columns and **c5** as the Sequence Column will result in::
+Selecting **c3** and **c4** as the **Title column(s)** and **c5** as the **Sequence column** will result in::
>902_419
GACTCATGATTTCTTACCTATTAGTGGTTGAACATC