update rmap and rmapq tool wrappers.

minor fixes of short read tool config.
This commit is contained in:
Wen-Yu Chung
2008-04-22 21:04:24 +00:00
parent d5253bf1b8
commit 4cdcab4715
12 changed files with 128 additions and 70 deletions
+1 -1
View File
@@ -2,7 +2,7 @@
<description>the percentage of reads supporting each nucleotide at each location</description>
<command interpreter="python">blat_coverage_report.py $input1 $output1</command>
<inputs>
<param name="input1" type="data" format="tabular" label="Alignment Result"/>
<param name="input1" type="data" format="tabular" label="Alignment result"/>
</inputs>
<outputs>
<data name="output1" format="tabular"/>
+1 -1
View File
@@ -2,7 +2,7 @@
<description>in wiggle format</description>
<command interpreter="python">blat_mapping.py $input1 $output1</command>
<inputs>
<param name="input1" type="data" format="tabular" label="Alignment Result"/>
<param name="input1" type="data" format="tabular" label="Alignment result"/>
</inputs>
<outputs>
<data name="output1" format="wig"/>
+6 -6
View File
@@ -8,7 +8,7 @@
</command>
<inputs>
<conditional name="source">
<param name="source_select" type="select" label="Target Source">
<param name="source_select" type="select" label="Target source">
<option value="database">Genome Build</option>
<option value="input_ref">Your Upload File</option>
</param>
@@ -16,13 +16,13 @@
<param name="dbkey" type="genomebuild" label="Genome" />
</when>
<when value="input_ref">
<param name="input_target" type="data" format="fasta" label="Reference Sequence" />
<param name="input_target" type="data" format="fasta" label="Reference sequence" />
</when>
</conditional>
<param name="input_query" type="data" format="fasta" label="Query Sequence"/>
<param name="iden" type="float" size="15" value="90.0" label="Minimal Identity (-minIdentity)" />
<param name="tile_size" type="integer" size="15" value="11" label="Minimal Size of Exact Match (-tileSize)" help="Must be between 6 and 18."/>
<param name="one_off" type="integer" size="15" value="0" label="Number of Mismatch in the Word (-oneOff)" help="Must be between 0 and 2." />
<param name="input_query" type="data" format="fasta" label="Sequence file"/>
<param name="iden" type="float" size="15" value="90.0" label="Minimal identity (-minIdentity)" />
<param name="tile_size" type="integer" size="15" value="11" label="Minimal size of exact match (-tileSize)" help="Must be between 6 and 18."/>
<param name="one_off" type="integer" size="15" value="0" label="Number of mismatch in the word (-oneOff)" help="Must be between 0 and 2." />
</inputs>
<outputs>
<data name="output1" format="tabular"/>
@@ -3,8 +3,8 @@
<command interpreter="python">convert_SOLiD_color2nuc.py $input1 $input2 $output1 </command>
<inputs>
<param name="input1" type="data" format="txt" label="SOLiD Color Coding File" />
<param name="input2" type="select" label="Keep Prefix Nucleotide">
<param name="input1" type="data" format="txt" label="SOLiD color coding file" />
<param name="input2" type="select" label="Keep prefix nucleotide">
<option value="yes">Yes</option>
<option value="no">No</option>
</param>
+2 -2
View File
@@ -2,10 +2,10 @@
<description>for Metagenomics Projects</description>
<command interpreter="python">megablast_wrapper.py $source_select $input_query $output1 $word_size $iden_cutoff $disc_word $disc_type $filter_query ${GALAXY_DATA_INDEX_DIR}</command>
<inputs>
<param name="input_query" type="data" format="fasta" label="Query Sequence"/>
<param name="source_select" type="select" display="radio" label="Target database">
<options from_file="blastdb.loc" name_col="0" value_col="0"/>
</param>
</param>
<param name="input_query" type="data" format="fasta" label="Sequence file"/>
<param name="word_size" type="select" label="Word size (-W)" help="Size of best perfect match">
<option value="28">28</option>
<option value="16">16</option>
+1 -1
View File
@@ -2,7 +2,7 @@
<description> </description>
<command interpreter="python">megablast_xml_parser.py $input1 $output1</command>
<inputs>
<param name="input1" type="data" format="txt" label="Megablast XML Output" />
<param name="input1" type="data" format="txt" label="Megablast XML output" />
</inputs>
<outputs>
<data name="output1" format="tabular"/>
+26 -15
View File
@@ -1,9 +1,5 @@
#! /usr/bin/python
"""
3. create test files, run functional test
"""
import os, sys, tempfile
assert sys.version_info[:2] >= (2.4)
@@ -24,22 +20,37 @@ def __main__():
mismatch = sys.argv[5] # -m
output_file = sys.argv[6]
# first guess the read length
guess_read_len = 0
seq = ''
for i, line in enumerate(open(infile)):
line = line.rstrip('\r\n')
if line.startswith('>'):
if seq:
guess_read_len = len(seq)
break
else:
seq += line
try:
test = int(read_len)
assert test >= 20 and test <= 64
if test == 0:
read_len = str(guess_read_len)
else:
assert test >= 20 and test <= 64
except:
stop_err('Invalid value for read length. Must be between 20 and 64.')
try:
int(align_len)
except:
stop_err('Invalid value for minimal length of an alignment.')
stop_err('Invalid value for minimal length of a hit.')
try:
test = int(mismatch)
assert test >= 0 and test <= int(0.1*int(read_len))
int(mismatch)
#assert test >= 0 and test <= int(0.1*int(read_len))
except:
stop_err('Invalid value for mismatch numbers in an alignment. Please use a value smaller than %d' % (int(0.1*int(read_len))))
stop_err('Invalid value for mismatch numbers in an alignment.')
all_files = []
if os.path.isdir(target_path):
@@ -59,14 +70,14 @@ def __main__():
command = "rmap -h %s -w %s -m %s -c %s %s -o %s 2>&1" % ( align_len, read_len, mismatch, detail_file_path, infile, output_tempfile )
#print command
try:
assert os.system( command ) == 0
except:
stop_err('Execution failed. Please check whether RMAP was installed.')
os.system( command )
except Exception, e:
stop_err( str( e ) )
try:
assert os.system( 'cat %s >> %s' % ( output_tempfile, output_file ) ) == 0
except:
stop_err('Failed to integrate files.')
os.system( 'cat %s >> %s' % ( output_tempfile, output_file ) )
except Exception, e:
stop_err( str( e ) )
try:
os.remove( output_tempfile )
+27 -9
View File
@@ -1,15 +1,33 @@
<tool id="rmap_wrapper" name="RMAP" version="1.0.0">
<description>for Solexa Short Reads Alignment</description>
<command interpreter="python">rmap_wrapper.py $database $input_seq $read_len $align_len $mismatch $output1
<command interpreter="python">
#if $trim.choice=="No": #rmap_wrapper.py $database $input_seq 0 $align_len $mismatch $output1
#else: #rmap_wrapper.py $database $input_seq $trim.read_len $align_len $mismatch $output1
#end if
</command>
<inputs>
<param name="database" type="select" display="radio" label="Target database">
<options from_file="faseq.loc" name_col="0" value_col="1"/>
</param>
<param name="input_seq" type="data" format="fasta" label="Query Sequence"/>
<param name="read_len" type="integer" size="15" value="36" label="Length of the Reads"/>
<param name="align_len" type="integer" size="15" value="36" label="Length of an Alignment"/>
<param name="mismatch" type="integer" size="15" value="3" label="Number of Mismatches Allowed"/>
<param name="input_seq" type="data" format="fasta" label="Sequence file"/>
<param name="align_len" type="integer" size="15" value="11" label="Minimal length of a hit (-h)" help="seed" />
<param name="mismatch" type="select" label="Number of mismatches allowed (-m)">
<option value="0">0</option>
<option value="1">1</option>
<option value="3">3</option>
<option value="5">5</option>
</param>
<conditional name="trim">
<param name="choice" type="select" label="To trim the reads">
<option value="No">No</option>
<option value="Yes">Yes</option>
</param>
<when value="No">
</when>
<when value="Yes">
<param name="read_len" type="integer" size="15" value="36" label="Read length (-w)"/>
</when>
</conditional>
</inputs>
<outputs>
<data name="output1" format="bed"/>
@@ -18,7 +36,7 @@
<tests>
<test>
<param name="database" value="/depot/data2/galaxy/faseq/test" />
<param name="input_seq" value="rmap_wrapper_test1.fa" ftype="fasta"/>
<param name="input_seq" value="rmap_wrapper_test1.fasta" ftype="fasta"/>
<param name="read_len" value="36" />
<param name="align_len" value="36" />
<param name="mismatch" value="3" />
@@ -34,7 +52,7 @@
.. class:: infomark
**TIP**. Each read should have at least the minimal length specified as the *Length of the Reads* parameter. Reads with lengths longer than the expected value will be trimmed at the 3'end.
**TIP**. The tool will guess the length of the reads, however, if you select to trim the reads, the *Reads length* must be between 20 and 64. Reads with lengths longer than the specified value will be trimmed at the 3'end.
-----
@@ -46,9 +64,9 @@ This tool runs **rmap** (for more information, please see the reference below),
**Parameters**
- *Length of the Reads* (**-w**) : the minimal length of the reads
- *Length of an Alignment* (**-h**) : the minimal length of an alignment
- *Minimal Length of a Hit* (**-h**) : this is the seed length or the minimal exact match length
- *Number of Mismatches Allowed* (**-m**) : the maximal number of mismatches allowed in an alignment
- *Read Length* (**-w**) : maximal length of the reads; reads longer than the threshold will be truncated at 3' end.
-----
+26 -15
View File
@@ -1,9 +1,5 @@
#! /usr/bin/python
"""
3. create test files, run functional test
"""
import os, sys, tempfile
assert sys.version_info[:2] >= (2.4)
@@ -36,23 +32,38 @@ def __main__():
int(high_len)
except:
stop_err('Invalid value for minimal high quality bases.')
# first guess the read length
guess_read_len = 0
seq = ''
for i, line in enumerate(open(infile)):
line = line.rstrip('\r\n')
if line.startswith('>'):
if seq:
guess_read_len = len(seq)
break
else:
seq += line
try:
test = int(read_len)
assert test >= 20 and test <= 64
if test == 0:
read_len = str(guess_read_len)
else:
assert test >= 20 and test <= 64
except:
stop_err('Invalid value for read length. Must be between 20 and 64.')
try:
int(align_len)
except:
stop_err('Invalid value for minimal length of an alignment.')
stop_err('Invalid value for minimal length of a hit.')
try:
test = int(mismatch)
assert test >= 0 and test <= int(0.1*int(read_len))
int(mismatch)
except:
stop_err('Invalid value for mismatch numbers in an alignment. Please use a number smaller than %d.' %(int(0.1*int(read_len))))
stop_err('Invalid value for mismatch numbers in an alignment.')
all_files = []
if os.path.isdir(target_path):
@@ -71,14 +82,14 @@ def __main__():
command = "rmapq -q %s -M %s -h %s -w %s -m %s -Q %s -c %s %s -o %s 2>&1" % ( high_score, high_len, align_len, read_len, mismatch, scorefile, detail_file_path, infile, output_tempfile )
#print command
try:
assert os.system( command ) == 0
except:
stop_err('Execution failed. Please check whether RMAP was installed.')
os.system( command )
except Exception, e:
stop_err( str( e ) )
try:
assert os.system( 'cat %s >> %s' % ( output_tempfile, output_file ) ) == 0
except:
stop_err('Failed to integrate files.')
except Exception, e:
stop_err( str( e ) )
try:
os.remove( output_tempfile )
+31 -13
View File
@@ -1,18 +1,36 @@
<tool id="rmapq_wrapper" name="RMAPQ" version="1.0.0">
<description>for Solexa Short Reads Alignment with Quality Scores</description>
<command interpreter="python">rmapq_wrapper.py $database $input_seq $input_score $high_score $high_len $read_len $align_len $mismatch $output1
<command interpreter="python">
#if $trim.choice=="No": #rmapq_wrapper.py $database $input_seq $input_score $high_score $high_len 0 $align_len $mismatch $output1
#else: #rmapq_wrapper.py $database $input_seq $input_score $high_score $high_len $trim.read_len $align_len $mismatch $output1
#end if
</command>
<inputs>
<param name="database" type="select" display="radio" label="Target database">
<options from_file="faseq.loc" name_col="0" value_col="1"/>
</param>
<param name="input_seq" type="data" format="fasta" label="Query Sequence"/>
<param name="input_score" type="data" format="qualityscore" label="Query Quality Score"/>
<param name="high_score" type="float" size="15" value="40" label="Minimum Score for High-quality Base"/>
<param name="high_len" type="integer" size="15" value="36" label="Minimal High-quality Bases"/>
<param name="read_len" type="integer" size="15" value="36" label="Length of the Reads"/>
<param name="align_len" type="integer" size="15" value="36" label="Length of an Alignment"/>
<param name="mismatch" type="integer" size="15" value="3" label="Number of Mismatches Allowed"/>
<param name="input_seq" type="data" format="fasta" label="Sequence file"/>
<param name="input_score" type="data" format="qualityscore" label="Quality score file"/>
<param name="high_score" type="float" size="15" value="40" label="Minimum score for high-quality base (-q)"/>
<param name="high_len" type="integer" size="15" value="36" label="Minimal high-quality bases (-M)"/>
<param name="align_len" type="integer" size="15" value="11" label="Minimal length of a hit (-h)" help="seed"/>
<param name="mismatch" type="select" label="Number of mismatches allowed (-m)">
<option value="0">0</option>
<option value="1">1</option>
<option value="3">3</option>
<option value="5">5</option>
</param>
<conditional name="trim">
<param name="choice" type="select" label="To trim the reads">
<option value="No">No</option>
<option value="Yes">Yes</option>
</param>
<when value="No">
</when>
<when value="Yes">
<param name="read_len" type="integer" size="15" value="36" label="Read length (-w)" />
</when>
</conditional>
</inputs>
<outputs>
<data name="output1" format="bed"/>
@@ -21,7 +39,7 @@
<tests>
<test>
<param name="database" value="/depot/data2/galaxy/faseq/test" />
<param name="input_seq" value="rmapq_wrapper_test1.fa" ftype="fasta"/>
<param name="input_seq" value="rmapq_wrapper_test1.fasta" ftype="fasta"/>
<param name="input_score" value="rmapq_wrapper_test1.qualityscore" ftype="qualityscore" />
<param name="high_score" value="40" />
<param name="high_len" value="36" />
@@ -36,11 +54,11 @@
.. class:: warningmark
RMAP was developed for **Solexa** reads.
RMAPQ was developed for **Solexa** reads.
.. class:: infomark
**TIP**. Each read should have at least the minimal length specified in the *Length of the Reads* parameter. Reads with lengths longer than the expected value will be trimmed at the 3'end.
**TIP**. The tool will guess the length of the reads, however, if you select to trim the reads, the *Maximal Length of the Reads* must be between 20 and 64. Reads with lengths longer than the specified value will be trimmed at the 3'end.
-----
@@ -54,9 +72,9 @@ This tool runs **rmapq** (for more information, please see the reference below),
- *Minimal High-quality Bases* (**-M**): the minimal length of the high quality score bases
- *Minimum Score for High-quality Base* (**-q**) : the minimal quality score
- *Length of the Reads* (**-w**) : the minimal length of the reads
- *Length of an Alignment* (**-h**) : the minimal length of an exact match (rmapq) or an alignment (rmap)
- *Minimal Length of a Hit* (**-h**) : the minimal length of an exact match or seed
- *Number of Mismatches Allowed* (**-m**) : the maximal number of mismatches allowed in an alignment
- *Read Length* (**-w**) : maximal length of the reads; reads longer than the threshold will be truncated at 3' end.
-----
@@ -5,8 +5,8 @@
<inputs>
<page>
<param name="input1" type="data" format="qualityscore,txtseq.zip" label="Sequence file" />
<param name="input2" type="integer" size="5" value="20" label="Quality Score Threshold" />
<param name="input1" type="data" format="qualityscore,txtseq.zip" label="Quality score file" help="No dataset? Read tip below"/>
<param name="input2" type="integer" size="5" value="20" label="Quality score threshold" />
</page>
</inputs>
<outputs>
@@ -29,7 +29,7 @@
.. class:: warningmark
To use this tool your dataset needs to be in *Quality Score* format. Click pencil icon next to your dataset to set datatype to *Quality Score*.
To use this tool your dataset needs to be in *Quality Score* format. Click pencil icon next to your dataset to set datatype to *Quality Score* (see below for examples of quality scores).
-----
+2 -2
View File
@@ -9,13 +9,13 @@
<inputs>
<page>
<conditional name="sequencing_method_choice">
<param name="sequencer" type="select" label="short read sequencing method">
<param name="sequencer" type="select" label="Short read sequencing method">
<option value="454">454 and SOLiD</option>
<option value="Solexa">Solexa</option>
</param>
<when value="454">
<param name="input1" type="data" format="fasta,txtseq.zip" label="Sequence file" />
<param name="input2" type="data" format="qualityscore,txtseq.zip" label="Score file" />
<param name="input2" type="data" format="qualityscore,txtseq.zip" label="Quality score file" />
<param name="input3" type="select" label="To keep homopolymers" >
<option value="yes">Yes</option>
<option value="no">No</option>