diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 1a5695c9d3b..997714a59c0 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -98,8 +98,8 @@ class FastqSolexa( Sequence ): dataset.peek = data.get_file_peek( dataset.file_name ) count = size = 0 bases_regexp = re.compile("^[NGTAC]*$") - for line in file( dataset.file_name ): - if line and line[0] == "@": + for i, line in enumerate(file( dataset.file_name )): + if line and line[0] == "@" and i % 4 == 0: count += 1 elif bases_regexp.match(line): line = line.strip() diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample index 7d4531a93ec..4dd0eea6032 100644 --- a/tool_conf.xml.sample +++ b/tool_conf.xml.sample @@ -274,6 +274,7 @@ +
diff --git a/tools/metag_tools/shrimp_wrapper.py b/tools/metag_tools/shrimp_wrapper.py index 98517ec077d..9feac6cc805 100644 --- a/tools/metag_tools/shrimp_wrapper.py +++ b/tools/metag_tools/shrimp_wrapper.py @@ -162,6 +162,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe readname, endindex = line[1:].split('/') else: score = line + if score: # the last one if hits.has_key(readname): if len(hits[readname]) == hit_per_read: @@ -182,8 +183,9 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe match_count = 0 if hit_per_read == 1: - matches = [ hits[readkey]['1'] ] - match_count = 1 + if len(hits[readkey]['1']) == 1: + matches = [ hits[readkey]['1'] ] + match_count = 1 else: end1_data = hits[readkey]['1'] end2_data = hits[readkey]['2'] @@ -591,6 +593,7 @@ def __main__(): if os.path.exists(query_qual_end2): os.remove(query_qual_end2) if os.path.exists(shrimp_log): os.remove(shrimp_log) + if __name__ == '__main__': __main__() diff --git a/tools/metag_tools/split_paired_reads.py b/tools/metag_tools/split_paired_reads.py new file mode 100644 index 00000000000..f2e4039e5a6 --- /dev/null +++ b/tools/metag_tools/split_paired_reads.py @@ -0,0 +1,46 @@ +#! /usr/bin/python + +""" +Split Solexa paired end reads +""" + +import os, sys + +if __name__ == '__main__': + + infile = sys.argv[1] + outfile_end1 = open(sys.argv[2], 'w') + outfile_end2 = open(sys.argv[3], 'w') + + for i, line in enumerate(file(infile)): + line = line.rstrip() + if not line or line.startswith('#'): continue + + end1 = '' + end2 = '' + + line_index = i % 4 + + if line_index == 0: + end1 = line + '/1' + end2 = line + '/2' + + elif line_index == 1: + seq_len = len(line)/2 + end1 = line[0:seq_len] + end2 = line[seq_len:] + + elif line_index == 2: + end1 = line + '/1' + end2 = line + '/2' + + else: + qual_len = len(line)/2 + end1 = line[0:qual_len] + end2 = line[qual_len:] + + outfile_end1.write('%s\n' %(end1)) + outfile_end2.write('%s\n' %(end2)) + + outfile_end1.close() + outfile_end2.close() \ No newline at end of file diff --git a/tools/metag_tools/split_paired_reads.xml b/tools/metag_tools/split_paired_reads.xml new file mode 100644 index 00000000000..f5c944c2bff --- /dev/null +++ b/tools/metag_tools/split_paired_reads.xml @@ -0,0 +1,56 @@ + + paired-end reads into two ends + + split_paired_reads.py $input $output1 $output2 + + + + + + + + + + + + + + + + +**What it does** + +This tool splits a single paired-end file in half and returns two files with each ends. + +----- + +**Input formats** + +A multiple-fastq file, for example:: + + @HWI-EAS91_1_30788AAXX:7:21:1542:1758 + GTCAATTGTACTGGTCAATACTAAAAGAATAGGATCGCTCCTAGCATCTGGAGTCTCTATCACCTGAGCCCA + +HWI-EAS91_1_30788AAXX:7:21:1542:1758 + hhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhh`hfhhVZSWehR + + +----- + +**Outputs** + +One end:: + + @HWI-EAS91_1_30788AAXX:7:21:1542:1758/1 + GTCAATTGTACTGGTCAATACTAAAAGAATAGGATC + +HWI-EAS91_1_30788AAXX:7:21:1542:1758/1 + hhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhhh + +The other end:: + + @HWI-EAS91_1_30788AAXX:7:21:1542:1758/2 + GCTCCTAGCATCTGGAGTCTCTATCACCTGAGCCCA + +HWI-EAS91_1_30788AAXX:7:21:1542:1758/2 + hhhhhhhhhhhhhhhhhhhhhhhh`hfhhVZSWehR + + +