-
-
Introducing Galactic Quickies
-
- Galactic quickies are super-short screencasts that are always under 5 minutes. We thought it may be a good way to spread the word about Galaxy's functionality while keeping the "annoyance factor" to the minimum. The quickies will be updated weekly.
-
-
-
-
-
Previous Quickies
+
Galaxy Quickies
diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample
index 9b8eae0f1c5..4b7bca75abf 100644
--- a/tool_conf.xml.sample
+++ b/tool_conf.xml.sample
@@ -185,7 +185,7 @@
-
+
diff --git a/tools/fastx_toolkit/fastq_to_fasta.xml b/tools/fastx_toolkit/fastq_to_fasta.xml
index ccfb30453d0..7db5e852793 100644
--- a/tools/fastx_toolkit/fastq_to_fasta.xml
+++ b/tools/fastx_toolkit/fastq_to_fasta.xml
@@ -1,6 +1,6 @@
converter
- gunzip -cf $input | fastq_to_fasta $SKIPN $RENAMESEQ -o $output -v
+ gunzip -cf $input | fastq_to_fasta -Q 33 $SKIPN $RENAMESEQ -o $output -v
diff --git a/tools/fastx_toolkit/fastx_collapser.xml b/tools/fastx_toolkit/fastx_collapser.xml
index e879ad265cd..7c16e8855a9 100644
--- a/tools/fastx_toolkit/fastx_collapser.xml
+++ b/tools/fastx_toolkit/fastx_collapser.xml
@@ -1,6 +1,6 @@
sequences
- zcat -f '$input' | fastx_collapser -v -o '$output'
+ zcat -f '$input' | fastx_collapser -Q 33 -v -o '$output'
diff --git a/tools/next_gen_conversion/solid2fastq.py b/tools/next_gen_conversion/solid2fastq.py
new file mode 100644
index 00000000000..6bddab5d034
--- /dev/null
+++ b/tools/next_gen_conversion/solid2fastq.py
@@ -0,0 +1,210 @@
+#!/usr/bin/env python
+
+import sys
+import string
+import optparse
+import tempfile
+import sqlite3
+
+def stop_err( msg ):
+ sys.stderr.write( msg )
+ sys.exit()
+
+def solid2sanger( quality_string, min_qual = 0 ):
+ sanger = ""
+ quality_string = quality_string.rstrip( " " )
+ for qv in quality_string.split(" "):
+ try:
+ if int( qv ) < 0:
+ qv = '0'
+ if int( qv ) < min_qual:
+ return False
+ break
+ sanger += chr( int( qv ) + 33 )
+ except:
+ pass
+ return sanger
+
+def Translator(frm='', to='', delete='', keep=None):
+ allchars = string.maketrans('','')
+ if len(to) == 1:
+ to = to * len(frm)
+ trans = string.maketrans(frm, to)
+ if keep is not None:
+ delete = allchars.translate(allchars, keep.translate(allchars, delete))
+ def callable(s):
+ return s.translate(trans, delete)
+ return callable
+
+def merge_reads_qual( f_reads, f_qual, f_out, trim_name=False, out='fastq', double_encode = False, trim_first_base = False, pair_end_flag = '', min_qual = 0, table_name=None ):
+
+ # Reads from two files f_csfasta (reads) and f_qual (quality values) and produces output in three formats depending on out parameter,
+ # which can have three values: fastq, txt, and db
+ # fastq = fastq format
+ # txt = space delimited format with defline, reads, and qvs
+ # dp = dump data into sqlite3 db.
+ # IMPORTNAT! If out = db two optins must be provided:
+ # 1. f_out must be a db connection object initialized with sqlite3.connect()
+ # 2. table_name must be provided
+
+ if out == 'db':
+ cursor = f_out.cursor()
+ sql = "create table %s (name varchar(50) not null, read blob, qv blob)" % table_name
+ cursor.execute(sql)
+
+ lines = []
+ line = " "
+ while line:
+ for f in [ f_reads, f_qual ]:
+ line = f.readline().rstrip( '\n\r' )
+ while line.startswith( '#' ):
+ line = f.readline().rstrip( '\n\r' )
+ lines.append( line )
+
+ if lines[0].startswith( '>' ):
+ defline = lines[0][1:]
+ if trim_name and ( defline[ len( defline )-3: ] == "_F3" or defline[ len( defline )-3: ] == "_R3" ):
+ defline = defline[ : len( defline )-3 ]
+
+ else:
+
+ if trim_first_base:
+ lines[0] = lines[0][1:]
+ if double_encode:
+ de = Translator(frm="0123.", to="ACGTN")
+ lines[0] = de(lines[0])
+ qual = solid2sanger( lines[1], int( min_qual ) )
+ if qual:
+ if out == 'fastq':
+ f_out.write( "@%s%s\n%s\n+\n%s\n" % ( defline, pair_end_flag, lines[0], qual ) )
+ if out == 'txt':
+ f_out.write( '%s %s %s\n' % (defline, lines[0], qual ) )
+ if out == 'db':
+ cursor.execute('insert into %s values("%s","%s","%s")' % (table_name, defline, lines[0], qual ) )
+ lines = []
+
+def main():
+
+ usage = "%prog --fr F3.csfasta --fq R3.csfasta --fout fastq_output_file [option]"
+ parser = optparse.OptionParser(usage=usage)
+
+
+ parser.add_option(
+ '--fr','--f_reads',
+ metavar="F3_CSFASTA_FILE",
+ dest='fr',
+ help='Name of F3 file with color space reads')
+
+ parser.add_option(
+ '--fq','--f_qual',
+ metavar="F3_QUAL_FILE",
+ dest='fq',
+ help='Name of F3 file with color quality values')
+
+ parser.add_option(
+ '--fout','--f3_fastq_output',
+ metavar="F3_OUTPUT",
+ dest='fout',
+ help='Name for F3 output file')
+
+ parser.add_option(
+ '--rr','--r_reads',
+ metavar="R3_CSFASTA_FILE",
+ dest='rr',
+ default = False,
+ help='Name of R3 file with color space reads')
+
+ parser.add_option(
+ '--rq','--r_qual',
+ metavar="R3_QUAL_FILE",
+ dest='rq',
+ default = False,
+ help='Name of R3 file with color quality values')
+
+ parser.add_option(
+ '--rout',
+ metavar="R3_OUTPUT",
+ dest='rout',
+ help='Name for F3 output file')
+
+ parser.add_option(
+ '-q','--min_qual',
+ dest='min_qual',
+ default = '-1000',
+ help='Minimum quality threshold for printing reads. If a read contains a single call with QV lower than this value, it will not be reported. Default is -1000')
+
+ parser.add_option(
+ '-t','--trim_name',
+ dest='trim_name',
+ action='store_true',
+ default = False,
+ help='Trim _R3 and _F3 off read names. Default is False')
+
+ parser.add_option(
+ '-f','--trim_first_base',
+ dest='trim_first_base',
+ action='store_true',
+ default = False,
+ help='Remove the first base of reads in color-space. Default is False')
+
+ parser.add_option(
+ '-d','--double_encode',
+ dest='de',
+ action='store_true',
+ default = False,
+ help='Double encode color calls as nucleotides: 0123. becomes ACGTN. Default is False')
+
+ options, args = parser.parse_args()
+
+ if not ( options.fout and options.fr and options.fq ):
+ parser.error("""
+ One or more of the three required paremetrs is missing:
+ (1) --fr F3.csfasta file
+ (2) --fq F3.qual file
+ (3) --fout name of output file
+ Use --help for more info
+ """)
+
+ fr = open ( options.fr , 'r' )
+ fq = open ( options.fq , 'r' )
+ f_out = open ( options.fout , 'w' )
+
+ if options.rr and options.rq:
+ rr = open ( options.rr , 'r' )
+ rq = open ( options.rq , 'r' )
+ if not options.rout:
+ parser.error("Provide the name for f3 output using --rout option. Use --help for more info")
+ r_out = open ( options.rout, 'w' )
+
+ db = tempfile.NamedTemporaryFile()
+ print db.name
+
+ try:
+ con = sqlite3.connect(db.name)
+ cur = con.cursor()
+ except:
+ stop_err('Cannot connect to %s\n') % db.name
+
+
+ merge_reads_qual( fr, fq, con, trim_name=options.trim_name, out='db', double_encode=options.de, trim_first_base=options.trim_first_base, min_qual=options.min_qual, table_name="f3" )
+ merge_reads_qual( rr, rq, con, trim_name=options.trim_name, out='db', double_encode=options.de, trim_first_base=options.trim_first_base, min_qual=options.min_qual, table_name="r3" )
+ cur.execute('create index f3_name on f3( name )')
+ cur.execute('create index r3_name on r3( name )')
+
+ cur.execute('select * from r3,f3 where f3.name = r3.name')
+ for item in cur:
+ f_out.write( "@%s%s\n%s\n+\n%s\n" % (item[0], "/1", item[1], item[2]) )
+ r_out.write( "@%s%s\n%s\n+\n%s\n" % (item[3], "/2", item[4], item[5]) )
+
+
+ else:
+ merge_reads_qual( fr, fq, f_out, trim_name=options.trim_name, out='fastq', double_encode = options.de, trim_first_base = options.trim_first_base, min_qual=options.min_qual )
+
+
+
+ f_out.close()
+
+if __name__ == "__main__":
+ main()
+
+
\ No newline at end of file
diff --git a/tools/next_gen_conversion/solid2fastq.xml b/tools/next_gen_conversion/solid2fastq.xml
new file mode 100644
index 00000000000..efe58f89715
--- /dev/null
+++ b/tools/next_gen_conversion/solid2fastq.xml
@@ -0,0 +1,159 @@
+
+ SOLiD output to fastq
+
+ #if $is_run.paired == "no" #solid2fastq.py --fr=$input1 --fq=$input2 --fout=$out_file1 -q $qual $trim_name $trim_first_base $double_encode
+ #elif $is_run.paired == "yes" #solid2fastq.py --fr=$input1 --fq=$input2 --fout=$out_file1 --rr=$input3 --rq=$input4 --rout=$out_file2 -q $qual $trim_name $trim_first_base $double_encode
+ #end if#
+
+
+
+
+
+
+ No
+ Yes
+
+
+
+
+
+
+
+
+
+
+ Yes
+ No
+
+
+ Yes
+ No
+
+
+ Yes
+ No
+
+
+
+
+
+ is_run['paired'] == 'yes'
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+**What it does**
+
+Converts output of SOLiD instrument (versions 3.5 and earlier) to fastq format suitable for bowtie, bwa, and PerM mappers.
+
+--------
+
+**Input datasets**
+
+Below are examples of forward (F3) reads and quality scores:
+
+Reads::
+
+ >1831_573_1004_F3
+ T00030133312212111300011021310132222
+ >1831_573_1567_F3
+ T03330322230322112131010221102122113
+
+Quality scores::
+
+ >1831_573_1004_F3
+ 4 29 34 34 32 32 24 24 20 17 10 34 29 20 34 13 30 34 22 24 11 28 19 17 34 17 24 17 25 34 7 24 14 12 22
+ >1831_573_1567_F3
+ 8 26 31 31 16 22 30 31 28 29 22 30 30 31 32 23 30 28 28 31 19 32 30 32 19 8 32 10 13 6 32 10 6 16 11
+
+
+**Mate pairs**
+
+If your data is from a mate-paired run, you will have one additional read and quality datasets that will look similar to the ones above with one exception: the names of reads will be ending with "_R3".
+In this case choose **Yes** from the *Is this a mate-pair run?* drop down and you will be able to select R reads. When processing mate pairs this tool generated two output files: one for F3 reads and the other for R3 reads.
+The reads are guaranteed to be paired -- mated reads will be in the same position in F3 and R3 fastq file. However, because pairing is verified it may take a while to process an entire SOLiD runs (several hours).
+
+------
+
+**Explanation of parameters**
+
+**Remove reads containing color qualities below this value** - any read that contains as least one color call with quality lower than the specified value **will not** be reported.
+
+**Trim trailing "_F3" and "_R3"?** - does just that. Not necessary for bowtie. Required for BWA.
+
+**Trim first base?** - SOLiD reads contain an adapter base such as the first T in this read::
+
+ >1831_573_1004_F3
+ T00030133312212111300011021310132222
+
+this option removes this base leaving only color calls. Not necessary for bowtie. Required for BWA.
+
+**Double encode?** - converts color calls (0123.) to pseudo-nucleotides (ACGTN). Not necessary for bowtie. Required for BWA.
+
+------
+
+**Examples of output**
+
+When all parameters are left "as-is" you will get this (using reads and qualities shown above)::
+
+ @1831_573_1004
+ T00030133312212111300011021310132222
+ +
+ %>CCAA9952+C>5C.?C79,=42C292:C(9/-7
+ @1831_573_1004
+ T03330322230322112131010221102122113
+ +
+ );@@17?@=>7??@A8?==@4A?A4)A+.'A+'1,
+
+Setting *Trim first base from reads* to **Yes** will produce this::
+
+ @1831_573_1004
+ 00030133312212111300011021310132222
+ +
+ %>CCAA9952+C>5C.?C79,=42C292:C(9/-7
+ @1831_573_1004
+ 03330322230322112131010221102122113
+ +
+ );@@17?@=>7??@A8?==@4A?A4)A+.'A+'1,
+
+Finally, setting *Double encode* to **Yes** will yield::
+
+ @1831_573_1004
+ TAAATACTTTCGGCGCCCTAAACCAGCTCACTGGGG
+ +
+ %>CCAA9952+C>5C.?C79,=42C292:C(9/-7
+ @1831_573_1004
+ TATTTATGGGTATGGCCGCTCACAGGCCAGCGGCCT
+ +
+ );@@17?@=>7??@A8?==@4A?A4)A+.'A+'1,
+
+
+
+
+
+
+