add fastq converters to implicit data type converter.

This commit is contained in:
Wen-Yu Chung
2008-05-28 19:13:37 +00:00
parent 8eb00bbd54
commit 0288543fe7
5 changed files with 200 additions and 0 deletions
+2
View File
@@ -2,6 +2,8 @@
<converters>
<converter file="bed_to_gff_converter.xml" source_datatype="bed" target_datatype="gff"/>
<converter file="fasta_to_tabular_converter.xml" source_datatype="fasta" target_datatype="tabular"/>
<converter file="fastq_to_fasta_converter.xml" source_datatype="fastq" target_datatype="fasta"/>
<converter file="fastq_to_qual_converter.xml" source_datatype="fastq" target_datatype="qual"/>
<converter file="gff_to_bed_converter.xml" source_datatype="gff" target_datatype="bed"/>
<converter file="interval_to_bed_converter.xml" source_datatype="interval" target_datatype="bed"/>
<converter file="maf_to_fasta_converter.xml" source_datatype="maf" target_datatype="fasta"/>
@@ -0,0 +1,69 @@
#! /usr/bin/python
"""
convert fastq file to separated sequence and quality files.
assume each sequence and quality score are contained in one line
the order should be:
1st line: @title_of_seq
2nd line: nucleotides
3rd line: +title_of_qualityscore (might be skipped)
4th line: quality scores
(in three forms: a. digits, b. ASCII codes, the first char as the coding base, c. ASCII codes without the first char.)
Usage:
%python convert_fastq2fasta.py <your_fastq_filename> <output_seq_filename> <output_score_filename>
"""
import sys, os
from math import *
assert sys.version_info[:2] >= ( 2, 4 )
def stop_err( msg ):
sys.stderr.write( "%s\n" % msg )
sys.exit()
if __name__ == '__main__':
# file I/O
infile = sys.argv[1]
outfile_seq = open(sys.argv[2], 'w')
# guessing the first char used in title lines
leading_char_seq_title = ''
every_four_lines = 0
for i, line in enumerate(file(infile)):
line = line.rstrip() # get rid of the newline and spaces
if ((not line) or (line.startswith('#'))): continue # comments
every_four_lines = (every_four_lines + 1) % 4
leading_char = line[0:1]
if every_four_lines == 1: # first line is expected to be read title
if not leading_char_seq_title:
leading_char_seq_title = leading_char
if leading_char != leading_char_seq_title:
stop_err('Invalid fastq format at line %d.' %(i))
read_title = line[1:]
outfile_seq.write('>%s\n' %(line[1:]))
elif every_four_lines == 2: # second line is expected to be read
read_length = len(line)
outfile_seq.write('%s\n' %(line))
else:
pass
outfile_seq.close()
@@ -0,0 +1,12 @@
<tool id="CONVERTER_fastq_to_fasta_0" name="FASTQ-to-FASTA" version="1.0.0">
<description>converts FASTQ file to FASTA format</description>
<command interpreter="python">fastq_to_fasta_converter.py $input $output</command>
<inputs>
<param name="input" type="data" format="fastq" label="Fastq file"/>
</inputs>
<outputs>
<data name="output" format="fasta"/>
</outputs>
<help>
</help>
</tool>
@@ -0,0 +1,106 @@
#! /usr/bin/python
"""
convert fastq file to separated sequence and quality files.
assume each sequence and quality score are contained in one line
the order should be:
1st line: @title_of_seq
2nd line: nucleotides
3rd line: +title_of_qualityscore (might be skipped)
4th line: quality scores
(in three forms: a. digits, b. ASCII codes, the first char as the coding base, c. ASCII codes without the first char.)
Usage:
%python convert_fastq2fasta.py <your_fastq_filename> <output_seq_filename> <output_score_filename>
"""
import sys, os
from math import *
assert sys.version_info[:2] >= ( 2, 4 )
def stop_err( msg ):
sys.stderr.write( "%s\n" % msg )
sys.exit()
if __name__ == '__main__':
# file I/O
infile = sys.argv[1]
outfile_score = open(sys.argv[2], 'w')
# guessing the first char used in title lines
leading_char_quality_title = ''
leading_char_seq_title = ''
default_coding_value = 64
every_four_lines = 0
for i, line in enumerate(file(infile)):
line = line.rstrip() # get rid of the newline and spaces
if ((not line) or (line.startswith('#'))): continue # comments
every_four_lines = (every_four_lines + 1) % 4
leading_char = line[0:1]
if every_four_lines == 1: # first line is expected to be read title
if not leading_char_seq_title:
leading_char_seq_title = leading_char
if leading_char != leading_char_seq_title:
stop_err('Invalid fastq format at line %d.' %(i))
read_title = line[1:]
elif every_four_lines == 2: # second line is expected to be read
read_length = len(line)
elif every_four_lines == 3: # third line is expected to be quality title
if not leading_char_quality_title:
leading_char_quality_title = leading_char
if leading_char != leading_char_quality_title:
stop_err('Invalid fastq format at line %d.' %(i))
quality_title = line[1:]
if (quality_title and (read_title != quality_title)):
stop_err('Invalid fastq format: titles for sequence and quality score are different.')
if not quality_title:
outfile_score.write('>%s\n' %(read_title))
else:
outfile_score.write('>%s\n' %(line[1:]))
else: # fourth line is expected to be the ASCII-coded quality scores
qual = ''
# peek: ascii code or digits?
first_value = line.split()[0]
if first_value.isdigit():
# digits
qual = line
else:
# ascii code
# guess leading char
quality_score_length = len(line)
if quality_score_length == (read_length+1): # first char is leading_char_score
leading_char_score = ord(line[0:1])
line = line[1:]
elif quality_score_length == read_length:
leading_char_score = default_coding_value # default
else:
stop_err('Invalid fastq format: the number of quality scores is not the same as bases.')
for j, char in enumerate(line):
score = ord(char)-leading_char_score # 64
qual += (str(score) + ' ')
outfile_score.write('%s\n' %(qual))
outfile_score.close()
@@ -0,0 +1,11 @@
<tool id="CONVERTER_fastq_to_qual_0" name="FASTQ-to-FASTA">
<command interpreter="python">fastq_to_qual_converter.py $input1 $output1 </command>
<inputs>
<param format="fastq" name="input1" type="data" label="Fastq file"/>
</inputs>
<outputs>
<data format="qual" name="output1" />
</outputs>
<help>
</help>
</tool>