From a19ae79b853d5b5b671f69ce6a2447898752e8de Mon Sep 17 00:00:00 2001 From: Daniel Blankenberg Date: Tue, 2 Mar 2010 10:47:23 -0500 Subject: [PATCH] Change color space FASTA file type from fastqsolid to fastqcssanger. Cripple accepted tool input formats for many of the FASTQ tools to only allow only fastqsanger and fastqcssanger to be used. --- datatypes_conf.xml.sample | 2 +- lib/galaxy/datatypes/sequence.py | 6 +-- lib/galaxy_utils/sequence/fasta.py | 2 +- lib/galaxy_utils/sequence/fastq.py | 6 +-- tool_conf.xml.sample | 19 +++++---- tools/fastq/fastq_combiner.py | 2 +- tools/fastq/fastq_combiner.xml | 15 +++---- tools/fastq/fastq_filter.xml | 32 +++++++-------- tools/fastq/fastq_groomer.xml | 50 +++++++++++------------ tools/fastq/fastq_manipulation.xml | 19 +++++++-- tools/fastq/fastq_paired_end_joiner.xml | 4 +- tools/fastq/fastq_paired_end_splitter.xml | 2 +- tools/fastq/fastq_stats.xml | 12 +----- tools/fastq/fastq_to_fasta.xml | 2 +- tools/fastq/fastq_trimmer.xml | 12 +++--- 15 files changed, 96 insertions(+), 89 deletions(-) diff --git a/datatypes_conf.xml.sample b/datatypes_conf.xml.sample index ef93ffe358c..b27c14128a1 100644 --- a/datatypes_conf.xml.sample +++ b/datatypes_conf.xml.sample @@ -35,7 +35,7 @@ - + diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 8dc665ebb8d..9a198b90d66 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -224,9 +224,9 @@ class FastqIllumina( Fastq ): """Class representing a FASTQ sequence ( the Illumina 1.3+ variant )""" file_ext = "fastqillumina" -class FastqSolid( Fastq ): - """Class representing a FASTQ sequence ( the SOLiD (color space) variant )""" - file_ext = "fastqsolid" +class FastqCSSanger( Fastq ): + """Class representing a Color Space FASTQ sequence ( e.g a SOLiD variant )""" + file_ext = "fastqcssanger" try: from galaxy import eggs diff --git a/lib/galaxy_utils/sequence/fasta.py b/lib/galaxy_utils/sequence/fasta.py index 08be930d64c..4917536c15b 100644 --- a/lib/galaxy_utils/sequence/fasta.py +++ b/lib/galaxy_utils/sequence/fasta.py @@ -104,7 +104,7 @@ class fastaWriter( object ): def __init__( self, fh ): self.file = fh def write( self, fastq_read ): - #this will include SOLiD adapter base if applicable + #this will include color space adapter base if applicable self.file.write( ">%s\n%s\n" % ( fastq_read.identifier[1:], fastq_read.sequence ) ) def close( self ): return self.file.close() diff --git a/lib/galaxy_utils/sequence/fastq.py b/lib/galaxy_utils/sequence/fastq.py index 552a4f6777b..d59fccf88fc 100644 --- a/lib/galaxy_utils/sequence/fastq.py +++ b/lib/galaxy_utils/sequence/fastq.py @@ -176,8 +176,8 @@ class fastqSolexaRead( fastqSequencingRead ): score_system = 'solexa' sequence_space = 'base' -class fastqSolidRead( fastqSequencingRead ): - format = 'solid' #color space +class fastqCSSangerRead( fastqSequencingRead ): + format = 'cssanger' #color space ascii_min = 33 ascii_max = 126 quality_min = 0 @@ -260,7 +260,7 @@ class fastqSolidRead( fastqSequencingRead ): FASTQ_FORMATS = {} -for format in [ fastqIlluminaRead, fastqSolexaRead, fastqSangerRead, fastqSolidRead ]: +for format in [ fastqIlluminaRead, fastqSolexaRead, fastqSangerRead, fastqCSSangerRead ]: FASTQ_FORMATS[ format.format ] = format diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample index 4337b15b2b2..922401d4238 100644 --- a/tool_conf.xml.sample +++ b/tool_conf.xml.sample @@ -175,30 +175,31 @@
-
diff --git a/tools/fastq/fastq_combiner.py b/tools/fastq/fastq_combiner.py index 3750e2b7714..790b096adf0 100644 --- a/tools/fastq/fastq_combiner.py +++ b/tools/fastq/fastq_combiner.py @@ -16,7 +16,7 @@ def main(): format = 'sanger' if fasta_type == 'csfasta' or qual_type == 'qualsolid': - format = 'solid' + format = 'cssanger' elif qual_type == 'qualsolexa': format = 'solexa' elif qual_type == 'qualillumina': diff --git a/tools/fastq/fastq_combiner.xml b/tools/fastq/fastq_combiner.xml index 6c953606ba3..d22fec3019c 100644 --- a/tools/fastq/fastq_combiner.xml +++ b/tools/fastq/fastq_combiner.xml @@ -1,4 +1,4 @@ - + into FASTQ fastq_combiner.py '$fasta_file' '${fasta_file.extension}' '$qual_file' '${qual_file.extension}' '$output_file' '$force_quality_encoding' @@ -13,8 +13,8 @@ - - + + @@ -25,13 +25,13 @@ - + - + @@ -55,7 +55,7 @@ - + @@ -63,7 +63,8 @@ This tool joins a FASTA file to a Quality Score file, creating a single FASTQ block for each read. -Specifying a set of quality scores is optional; when not provided, the output will be fastqsanger or fastqsolid (when a csfasta is provided) with each quality score being the maximal allowed value (93). +Specifying a set of quality scores is optional; when not provided, the output will be fastqsanger or fastqcssanger (when a csfasta is provided) with each quality score being the maximal allowed value (93). +Use this tool, for example, to convert 454-type output to FASTQ. diff --git a/tools/fastq/fastq_filter.xml b/tools/fastq/fastq_filter.xml index 10f9704a7ac..b4755b76b3c 100644 --- a/tools/fastq/fastq_filter.xml +++ b/tools/fastq/fastq_filter.xml @@ -3,7 +3,7 @@ fastq_filter.py $input_file $fastq_filter_file $output_file $output_file.files_path '${input_file.extension[len( 'fastq' ):]}' - + @@ -21,20 +21,20 @@ - + int( float( value ) ) == float( value ) - + int( float( value ) ) == float( value ) - + - + @@ -120,7 +120,7 @@ ret_val = fastq_read_pass_filter( fastq_read ) - + + --> + - + @@ -151,7 +151,7 @@ ret_val = fastq_read_pass_filter( fastq_read ) - + @@ -170,7 +170,7 @@ ret_val = fastq_read_pass_filter( fastq_read ) - + @@ -203,7 +203,7 @@ ret_val = fastq_read_pass_filter( fastq_read ) - + @@ -292,10 +292,10 @@ This tool allows you to build complex filters to be applied to each read in a FA Basic Options: * You can specify a minimum and maximum read lengths. * You can specify minimum and maximum per base quality scores, with optionally specifying the number of bases that are allowed to deviate from this range (default of 0 deviant bases). + * If your data is paired-end, select the proper checkbox; this will cause each read to be internally split down the middle and filters applied to each half using the offsets specified. Advance Options: * You can specify any number of advanced filters. - * If your data is paired-end, select the proper checkbox; this will cause each read to be internally split down the middle and filters applied to each half using the offsets specified. * Offsets are defined, starting at zero, increasing from the ends of the reads. For example, a quality string of "ABCDEFG", with offsets of 1 and 1 specified will yield "BCDEF". * You can specify either absolute offset values, or percentage offset values. When using the percent-based method, offsets are rounded to the nearest integer. * The user specifies the aggregating action (min, max, sum, mean) to perform on the quality score values found between the specified offsets to be used with the user defined comparison operation and comparison value. @@ -305,7 +305,7 @@ Advance Options: .. class:: warningmark -Adapter bases in SOLiD reads are excluded from filtering. +Adapter bases in color space reads are excluded from filtering. diff --git a/tools/fastq/fastq_groomer.xml b/tools/fastq/fastq_groomer.xml index 8df0c09576c..44305e7dbc5 100644 --- a/tools/fastq/fastq_groomer.xml +++ b/tools/fastq/fastq_groomer.xml @@ -1,4 +1,4 @@ - + convert between various FASTQ quality formats fastq_groomer.py '$input_file' '$input_type' '$output_file' '$output_type' '$force_quality_encoding' '$summarize_input' @@ -7,13 +7,13 @@ - + - + @@ -31,7 +31,7 @@ - + @@ -66,10 +66,10 @@ - + - + @@ -99,10 +99,10 @@ - + - + @@ -132,39 +132,39 @@ - + - + - + - - - + + + - + - - + + - - + + - - + + @@ -240,13 +240,13 @@ When converting, if a quality score falls outside of the target score range, it When converting between Solexa and the other formats, quality scores are mapped between Solexa and PHRED scales using the equations found in Cock PJ, Fields CJ, Goto N, Heuer ML, Rice PM. The Sanger FASTQ file format for sequences with quality scores, and the Solexa/Illumina FASTQ variants. Nucleic Acids Res. 2009 Dec 16. -When converting between color space (SOLiD) and base/sequence space (Sanger, Illumina, Solexa) formats, adapter bases are lost or gained; if gained, the base 'G' is used as the adapter. You cannot convert a color space read to base space if there is no adapter present in the color space sequence. Any masked or ambiguous nucleotides in base space will be converted to 'N's when determining color space encoding. +When converting between color space (csSanger) and base/sequence space (Sanger, Illumina, Solexa) formats, adapter bases are lost or gained; if gained, the base 'G' is used as the adapter. You cannot convert a color space read to base space if there is no adapter present in the color space sequence. Any masked or ambiguous nucleotides in base space will be converted to 'N's when determining color space encoding. ----- **Examples** -- Converting the Solexa FASTQ data:: +1. Converting the Solexa FASTQ data:: @Solexa scores from -5 to 62 inclusive (in that order) ACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGT @@ -279,7 +279,7 @@ When converting between color space (SOLiD) and base/sequence space (Sanger, Ill + ~}|{zyxwvutsrqponmlkjihgfedcba`_^]\[ZYXWVUTSRQPONMLKJJIHGFEEDDCCBBAA -- Converting the Illumina 1.3+ FASTQ data:: +2. Converting the Illumina 1.3+ FASTQ data:: @Illumina PHRED scores from 0 to 62 inclusive (in that order) ACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACG @@ -312,7 +312,7 @@ When converting between color space (SOLiD) and base/sequence space (Sanger, Ill + ~}|{zyxwvutsrqponmlkjihgfedcba`_^]\[ZYXWVUTSRQPONMLKJHGFECB@>;; -- Converting standard Sanger FASTQ:: +3. Converting standard Sanger FASTQ:: @Sanger PHRED scores from 0 to 93 inclusive (in that order) ACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTACGTAC diff --git a/tools/fastq/fastq_manipulation.xml b/tools/fastq/fastq_manipulation.xml index b68b4e1facf..73de81c41dd 100644 --- a/tools/fastq/fastq_manipulation.xml +++ b/tools/fastq/fastq_manipulation.xml @@ -5,7 +5,7 @@ - + @@ -376,13 +376,26 @@ def match_and_manipulate_read( fastq_read ): This tool allows you to build complex manipulations to be applied to each matching read in a FASTQ file. A read must match all matching directives in order for it to be manipulated; if a read does not match, it is output in a non-modified manner. All reads matching will have each of the specified manipulations performed upon them, in the order specified. Regular Expression Matches are made using re.search, see http://docs.python.org/library/re.html for more information. + All matching is performed on a single line string, regardless if e.g. the sequence or quality score spans multiple lines in the original file. String translations are performed using string.translate, see http://docs.python.org/library/string.html#string.translate and http://docs.python.org/library/string.html#string.maketrans for more information. - .. class:: warningmark -Only SOLiD (color space) reads can have adapter bases substituted. +Only color space reads can have adapter bases substituted. + +----- + +**Example** + +Suppose you have a color space sanger formatted sequence (fastqcssanger) and you want to double-encode the color space into psuedo-nucleotide space (this is different from converting) to allow these reads to be used in tools which do not natively support it (using specially designed indexes). This tool can handle this manipulation, however, this is generally not recommended as results tend to be poorer than those produced from tools which are specially designed to handle color space data. + +Steps: + +1. Click **Add new Match Reads** and leave the matching options set to the default (Matching by sequence name/identifier using the regular expression "\*."; thereby matching all reads). +2. Click **Add new Manipulate Reads**, change **Manipulate Reads on** to "Sequence Content", set **Sequence Manipulation Type** to "Change Adapter Base" and set **New Adapter** to "" (an empty text field). +3. Click **Add new Manipulate Reads**, change **Manipulate Reads on** to "Sequence Content", set **Sequence Manipulation Type** to "String Translate" and set **From** to "0123." and **To** to "ACGTN". +4. Click Execute. The new history item will contained double-encoded psuedo-nucleotide space reads. diff --git a/tools/fastq/fastq_paired_end_joiner.xml b/tools/fastq/fastq_paired_end_joiner.xml index e63c0be0cd2..eebf7ed1b9f 100644 --- a/tools/fastq/fastq_paired_end_joiner.xml +++ b/tools/fastq/fastq_paired_end_joiner.xml @@ -2,8 +2,8 @@ on paired end reads fastq_paired_end_joiner.py '$input1_file' '${input1_file.extension[len( 'fastq' ):]}' '$input2_file' '${input2_file.extension[len( 'fastq' ):]}' '$output_file' - - + + diff --git a/tools/fastq/fastq_paired_end_splitter.xml b/tools/fastq/fastq_paired_end_splitter.xml index c9cd409cd51..bc5e9a52d92 100644 --- a/tools/fastq/fastq_paired_end_splitter.xml +++ b/tools/fastq/fastq_paired_end_splitter.xml @@ -2,7 +2,7 @@ on joined paired end reads fastq_paired_end_splitter.py '$input1_file' '${input1_file.extension[len( 'fastq' ):]}' '$output1_file' '$output2_file' - + diff --git a/tools/fastq/fastq_stats.xml b/tools/fastq/fastq_stats.xml index 265e0740b45..e9183556acf 100644 --- a/tools/fastq/fastq_stats.xml +++ b/tools/fastq/fastq_stats.xml @@ -2,7 +2,7 @@ by column fastq_stats.py '$input_file' '$output_file' '${input_file.extension[len( 'fastq' ):]}' - + @@ -53,20 +53,12 @@ For example:: 3 14336356 2 34 433659182 30.2489127642 29.0 32.0 33.0 4.0 23 34 2,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22 4310988 2941988 3437467 3645784 129 4 14336356 2 34 433635331 30.2472490917 29.0 32.0 33.0 4.0 23 34 2,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22 4110637 3007028 3671749 3546839 103 5 14336356 2 34 432498583 30.167957813 29.0 32.0 33.0 4.0 23 34 2,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22 4348275 2935903 3293025 3759029 124 - 6 14336356 2 34 433516130 30.2389344963 29.0 32.0 33.0 4.0 23 34 2,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22 4148685 3240703 3395388 3551483 97 - 7 14336356 2 34 433280560 30.2225028452 29.0 32.0 33.0 4.0 23 34 2,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22 4080445 3143867 3181629 3930294 121 - 8 14336356 2 34 433253015 30.2205815062 29.0 32.0 33.0 4.0 23 34 2,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22 4021204 3164843 3257779 3892419 111 - 9 14336356 2 34 432416621 30.1622407396 29.0 32.0 33.0 4.0 23 34 2,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22 3952703 3229316 3263002 3891205 130 - 10 14336356 2 34 431795072 30.1188859986 29.0 32.0 33.0 4.0 23 34 2,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22 4139373 3227694 3262197 3706968 124 - 11 14336356 2 34 430636560 30.0380766214 29.0 32.0 33.0 4.0 23 34 2,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22 4078833 3194895 3305010 3757469 149 - 12 14336356 2 34 426265591 29.7331895916 29.0 32.0 33.0 4.0 23 34 2,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22 3980583 3224693 3288479 3842447 154 - 13 14336356 2 34 425042610 29.6478833254 28.0 32.0 33.0 5.0 21 34 2,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20 3994385 3201916 3236851 3903054 150 ----- .. class:: warningmark -Adapter bases in SOLiD reads are excluded from statistics. +Adapter bases in color space reads are excluded from statistics. diff --git a/tools/fastq/fastq_to_fasta.xml b/tools/fastq/fastq_to_fasta.xml index 883897b97ba..de8bf70dca8 100644 --- a/tools/fastq/fastq_to_fasta.xml +++ b/tools/fastq/fastq_to_fasta.xml @@ -15,7 +15,7 @@ - + diff --git a/tools/fastq/fastq_trimmer.xml b/tools/fastq/fastq_trimmer.xml index deb129fc7a3..1f50deef134 100644 --- a/tools/fastq/fastq_trimmer.xml +++ b/tools/fastq/fastq_trimmer.xml @@ -2,27 +2,27 @@ by column fastq_trimmer.py '$input_file' '$output_file' '${offset_type['left_column_offset']}' '${offset_type['right_column_offset']}' '${offset_type['base_offset_type']}' '${input_file.extension[len( 'fastq' ):]}' '$keep_zero_length' - + - + int( float( value ) ) == float( value ) - + int( float( value ) ) == float( value ) - + - + @@ -107,7 +107,7 @@ Or you set percent offsets of 6% and 20% (corresponds to absolute offsets of 2,7 .. class:: warningmark -Trimming a SOLiD read will cause any adapter base to be lost. +Trimming a color space read will cause any adapter base to be lost.