From a3ee9be7356e9be06c1cccdc2706b5f6aa5cfbab Mon Sep 17 00:00:00 2001 From: Anton Nekrutenko Date: Tue, 29 Apr 2008 14:36:04 +0000 Subject: [PATCH] More clean up and modification of FASTA tools --- tool_conf.xml.main | 7 +++++ tool_conf.xml.sample | 2 ++ tools/fasta_tools/fasta_compute_length.py | 6 +++- tools/fasta_tools/fasta_compute_length.xml | 31 +++++++++---------- .../fasta_concatenate_by_species.xml | 11 +++++-- tools/fasta_tools/fasta_filter_by_length.xml | 16 +++++----- tools/fasta_tools/fasta_to_tabular.py | 7 ++++- tools/fasta_tools/fasta_to_tabular.xml | 29 ++++++++++------- tools/fasta_tools/tabular_to_fasta.xml | 12 +++++-- 9 files changed, 77 insertions(+), 44 deletions(-) diff --git a/tool_conf.xml.main b/tool_conf.xml.main index ebbc28386fb..1064796a325 100644 --- a/tool_conf.xml.main +++ b/tool_conf.xml.main @@ -56,6 +56,13 @@ +
+ + + + + +
diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample index d5849300b20..3db546f2bb6 100644 --- a/tool_conf.xml.sample +++ b/tool_conf.xml.sample @@ -264,6 +264,8 @@ + +
diff --git a/tools/fasta_tools/fasta_compute_length.py b/tools/fasta_tools/fasta_compute_length.py index 650deda0b5f..9bd36225a37 100644 --- a/tools/fasta_tools/fasta_compute_length.py +++ b/tools/fasta_tools/fasta_compute_length.py @@ -12,10 +12,14 @@ assert sys.version_info[:2] >= ( 2, 4 ) def __main__(): input_filename = sys.argv[1] output_filename = sys.argv[2] + keep_first = int( sys.argv[3] ) + 1 tmp_title = tmp_seq = '' tmp_seq_count = 0 seq_hash = {} + if keep_first == 0: + keep_first = None + for i, line in enumerate( file( input_filename ) ): line = line.rstrip( '\r\n' ) if not line or line.startswith( '#' ): @@ -38,7 +42,7 @@ def __main__(): output_handle = open( output_filename, 'w' ) for i, fasta_title in title_keys: tmp_seq = seq_hash[ ( i, fasta_title ) ] - output_handle.write( "%s\t%d\n" % ( fasta_title[ 1: ], len( tmp_seq ) ) ) + output_handle.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( tmp_seq ) ) ) output_handle.close() if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/tools/fasta_tools/fasta_compute_length.xml b/tools/fasta_tools/fasta_compute_length.xml index 10e203f67c9..3778a7ccccd 100644 --- a/tools/fasta_tools/fasta_compute_length.xml +++ b/tools/fasta_tools/fasta_compute_length.xml @@ -1,8 +1,9 @@ - - - fasta_compute_length.py $input $output + + sequence length + fasta_compute_length.py $input $output $keep_first - + + @@ -10,10 +11,12 @@ + + @@ -21,27 +24,21 @@ **What it does** - This tool counts the length of each fasta sequence in the file. The output file has two columns per line (separated by tab): fasta titles and lengths of the sequences. +This tool counts the length of each fasta sequence in the file. The output file has two columns per line (separated by tab): fasta titles and lengths of the sequences. The option *How many characters to keep?* allows to select a specified number of letters from the beginning of each FASTA entry. ----- **Example** -- assume the input file contains fasta sequences:: +Suppose you have the following FASTA formatted sequences from a Roche (454) FLX sequencing run:: - >seq1 - TCATTTA - >seq2 - ATGGCGTCGGCC - >seq3 - TCACATGATG + >EYKX4VC02EQLO5 length=108 xy=1826_0455 region=2 run=R_2007_11_07_16_15_57_ TCCGCGCCGAGCATGCCCATCTTGGATTCCGGCGCGATGACCATCGCCCGCTCCACCACG TTCGGCCGGCCCTTCTCGTCGAGGAATGACACCAGCGCTTCGCCCACG >EYKX4VC02D4GS2 length=60 xy=1573_3972 region=2 run=R_2007_11_07_16_15_57_ AATAAAACTAAATCAGCAAAGACTGGCAAATACTCACAGGCTTATACAATACAAATGTAA -- the tool will return(first column is the titles, second column is the length of the sequences):: - - >seq1 7 - >seq2 12 - >seq3 10 +Running this tool while setting **How many characters to keep?** to **14** will produce this:: + EYKX4VC02EQLO5 108 + EYKX4VC02D4GS2 60 + \ No newline at end of file diff --git a/tools/fasta_tools/fasta_concatenate_by_species.xml b/tools/fasta_tools/fasta_concatenate_by_species.xml index f78aab51d93..f85976c3574 100644 --- a/tools/fasta_tools/fasta_concatenate_by_species.xml +++ b/tools/fasta_tools/fasta_concatenate_by_species.xml @@ -1,5 +1,5 @@ - - alignment by species + + FASTA alignment by species fasta_concatenate_by_species.py $input1 $out_file1 @@ -14,9 +14,14 @@ + +**What it does** + This tools attempts to parse FASTA headers to determine the species for each sequence in a multiple FASTA alignment. It then linearly concatenates the sequences for each species in the file, creating one sequence per determined species. +------- + **Example** Starting FASTA:: @@ -45,7 +50,7 @@ Starting FASTA:: -Becomes:: +becomes:: >hg18 GTGT diff --git a/tools/fasta_tools/fasta_filter_by_length.xml b/tools/fasta_tools/fasta_filter_by_length.xml index 27200919ab4..464d56db317 100644 --- a/tools/fasta_tools/fasta_filter_by_length.xml +++ b/tools/fasta_tools/fasta_filter_by_length.xml @@ -1,10 +1,10 @@ - - + + sequences by length fasta_filter_by_length.py $input $min_length $max_length $output - - + + @@ -21,19 +21,19 @@ .. class:: infomark -**TIP**. If only want to show sequences longer than a threshold, set *minimal length* to the threshold and leave *maximum length* to zero. +**TIP**. To return sequences longer than a certain length, set *Minimal length* to desired value and leave *Maximum length* set to '0'. ----- **What it does** - This tool accepts two parameters: *minimal length* and *maximum length*, and returns sequences of length within the two thresholds. +Outputs sequences between *Minimal length* and *Maximum length*. ----- **Example** -- assume the input file contains fasta sequences:: +Suppose you have the following FASTA formatted sequences:: >seq1 TCATTTAATGAC @@ -44,7 +44,7 @@ >seq4 ATGGAAGC -- return sequences with length longer than 10bp (set the *minimal length* to 10, and the *maximum length* to 0 (no limitation)):: +Setting the **Minimal length** to **10**, and the **Maximum length** to **0** will return all sequences longer than 10 bp:: >seq1 TCATTTAATGAC diff --git a/tools/fasta_tools/fasta_to_tabular.py b/tools/fasta_tools/fasta_to_tabular.py index 26e538f6a71..c1e8eaede1f 100644 --- a/tools/fasta_tools/fasta_to_tabular.py +++ b/tools/fasta_tools/fasta_to_tabular.py @@ -13,9 +13,14 @@ seq_hash = {} def __main__(): infile = sys.argv[1] outfile = sys.argv[2] + keep_first = int( sys.argv[3] ) + 1 title = '' sequence = '' sequence_count = 0 + + if keep_first == 0: + keep_first = None + for i, line in enumerate( open( infile ) ): line = line.rstrip( '\r\n' ) if not line or line.startswith( '#' ): @@ -38,7 +43,7 @@ def __main__(): out = open( outfile, 'w' ) for i, fasta_title in title_keys: sequence = seq_hash[( i, fasta_title )] - out.write( "%s\t%s\n" %( fasta_title, sequence ) ) + out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], sequence ) ) out.close() if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/tools/fasta_tools/fasta_to_tabular.xml b/tools/fasta_tools/fasta_to_tabular.xml index e0f5a742f70..4d8fdae6652 100644 --- a/tools/fasta_tools/fasta_to_tabular.xml +++ b/tools/fasta_tools/fasta_to_tabular.xml @@ -1,8 +1,9 @@ - Converts a FASTA file to Tabular format - fasta_to_tabular.py $input $output + converts FASTA file to tabular format + fasta_to_tabular.py $input $output $keep_first - + + @@ -10,28 +11,34 @@ + + +**What it does** + +This tool converts FASTA formatted sequences to TAB-delimited format. The option *How many characters to keep?* allows to select a specified number of letters from the beginning of each FASTA entry. + +----- + **Example** -A fasta file with two sequences:: +Suppose you have the following FASTA formatted sequences from a Roche (454) FLX sequencing run:: - >seq1 - CCGGTATCCG - >seq2 - CTTACC + >EYKX4VC02EQLO5 length=108 xy=1826_0455 region=2 run=R_2007_11_07_16_15_57_ TCCGCGCCGAGCATGCCCATCTTGGATTCCGGCGCGATGACCATCGCCCGCTCCACCACG TTCGGCCGGCCCTTCTCGTCGAGGAATGACACCAGCGCTTCGCCCACG >EYKX4VC02D4GS2 length=60 xy=1573_3972 region=2 run=R_2007_11_07_16_15_57_ AATAAAACTAAATCAGCAAAGACTGGCAAATACTCACAGGCTTATACAATACAAATGTAA -Returns:: +Running this tool while setting **How many characters to keep?** to **14** will produce this:: + + EYKX4VC02EQLO5 TCCGCGCCGAGCATGCCCATCTTGGATTCCGGCGCGATGACCATCGCCCGCTCCACCACGTTCGGCCGGCCCTTCTCGTCGAGGAATGACACCAGCGCTTCGCCCACG + EYKX4VC02D4GS2 AATAAAACTAAATCAGCAAAGACTGGCAAATACTCACAGGCTTATACAATACAAATGTAA - >seq1 CCGGTATCCG - >seq2 CTTACC \ No newline at end of file diff --git a/tools/fasta_tools/tabular_to_fasta.xml b/tools/fasta_tools/tabular_to_fasta.xml index 1e4a0e87700..75faaf1e715 100644 --- a/tools/fasta_tools/tabular_to_fasta.xml +++ b/tools/fasta_tools/tabular_to_fasta.xml @@ -1,5 +1,5 @@ - Converts a tabular file to FASTA format + converts tabular file to FASTA format tabular_to_fasta.py $input $title_col $seq_col $output @@ -19,14 +19,20 @@ +**What it does** + +Converts tab delimited data into FASTA formatted sequences. + +----------- + **Example** -Solexa data:: +Suppose this is a sequence file produced by Illumina (Solexa) sequencer:: 5 300 902 419 GACTCATGATTTCTTACCTATTAGTGGTTGAACATC 5 300 880 431 GTGATATGTATGTTGACGGCCATAAGGCTGCTTCTT -Selecting **c3 and c4** as the Title Columns and **c5** as the Sequence Column will result in:: +Selecting **c3** and **c4** as the **Title column(s)** and **c5** as the **Sequence column** will result in:: >902_419 GACTCATGATTTCTTACCTATTAGTGGTTGAACATC