mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
update blat and megablast wrapper. change blastdb.loc file, now don't need all chunks id.
This commit is contained in:
@@ -4,6 +4,6 @@
|
||||
#
|
||||
#TODO: fill in this format
|
||||
#
|
||||
nt /depot/data2/galaxy/blastdb/nt/nt.chunk.00 /depot/data2/galaxy/blastdb/nt/nt.chunk.01 /depot/data2/galaxy/blastdb/nt/nt.chunk.02 /depot/data2/galaxy/blastdb/nt/nt.chunk.03 /depot/data2/galaxy/blastdb/nt/nt.chunk.04 /depot/data2/galaxy/blastdb/nt/nt.chunk.05 /depot/data2/galaxy/blastdb/nt/nt.chunk.06 /depot/data2/galaxy/blastdb/nt/nt.chunk.07 /depot/data2/galaxy/blastdb/nt/nt.chunk.08 /depot/data2/galaxy/blastdb/nt/nt.chunk.09 /depot/data2/galaxy/blastdb/nt/nt.chunk.10 /depot/data2/galaxy/blastdb/nt/nt.chunk.11 /depot/data2/galaxy/blastdb/nt/nt.chunk.12 /depot/data2/galaxy/blastdb/nt/nt.chunk.13 /depot/data2/galaxy/blastdb/nt/nt.chunk.14 /depot/data2/galaxy/blastdb/nt/nt.chunk.15 /depot/data2/galaxy/blastdb/nt/nt.chunk.16 /depot/data2/galaxy/blastdb/nt/nt.chunk.17 /depot/data2/galaxy/blastdb/nt/nt.chunk.18 /depot/data2/galaxy/blastdb/nt/nt.chunk.19 /depot/data2/galaxy/blastdb/nt/nt.chunk.20 /depot/data2/galaxy/blastdb/nt/nt.chunk.21 /depot/data2/galaxy/blastdb/nt/nt.chunk.22 /depot/data2/galaxy/blastdb/nt/nt.chunk.23
|
||||
nt /depot/data2/galaxy/blastdb/nt/nt.chunk
|
||||
nr /depot/data2/galaxy/blastdb/nr/nr.chunk.0 /depot/data2/galaxy/blastdb/nr/nr.chunk.1 /depot/data2/galaxy/blastdb/nr/nr.chunk.2
|
||||
test /depot/data2/galaxy/blastdb/test/test.fa
|
||||
|
||||
@@ -62,6 +62,7 @@ def __main__():
|
||||
try:
|
||||
test = int(sys.argv[7])
|
||||
one_off = sys.argv[7]
|
||||
assert test >= 0 and test <= int(tile_size)
|
||||
except:
|
||||
stop_err('Invalid value for mismatch numbers in the word')
|
||||
|
||||
|
||||
@@ -22,7 +22,7 @@
|
||||
<param name="input_query" type="data" format="fasta" label="Query Sequence"/>
|
||||
<param name="iden" type="float" size="15" value="90.0" label="Minimal Identity (-minIdentity)" />
|
||||
<param name="tile_size" type="integer" size="15" value="11" label="Minimal Size of Exact Match (-tileSize)" help="for shorter reads, use size = 8. Must be between 6 and 18."/>
|
||||
<param name="one_off" type="integer" size="15" value="0" label="Number of Mismatch in the Word (-oneOff)" />
|
||||
<param name="one_off" type="integer" size="15" value="0" label="Number of Mismatch in the Word (-oneOff)" help="Must be between 0 and 2." />
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data name="output1" format="tabular"/>
|
||||
@@ -42,13 +42,17 @@
|
||||
|
||||
.. class:: warningmark
|
||||
|
||||
Use smaller word size (*Minimal Size of Exact Match*) will increase the computational time.
|
||||
Use a smaller word size (*Minimal Size of Exact Match*) will increase the computational time.
|
||||
|
||||
.. class:: warningmark
|
||||
|
||||
Use a larger mismatch number (*Number of Mismatch in the Word*) will increase the computational time.
|
||||
|
||||
-----
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool runs alignment program **BLAT**. Your short reads file is searched against a genome build (select from table) or another uploaded file.
|
||||
This tool runs alignment program **BLAT**. Your short reads file is searched against a genome build or another uploaded file.
|
||||
|
||||
-----
|
||||
|
||||
@@ -59,16 +63,29 @@ Use smaller word size (*Minimal Size of Exact Match*) will increase the computat
|
||||
>seq1
|
||||
TGGTAATGGTGGTTTTTTTTTTTTTTTTTTATTTTT
|
||||
|
||||
- Search against ce2 (C. elegans March 2004)::
|
||||
- Use default settings.
|
||||
|
||||
- Search against ce2 (C. elegans March 2004), partial result::
|
||||
|
||||
25 1 0 0 0 0 0 0 + seq1 36 10 36 chrI 15080483 9704438 9704464 1 26, 10, 9704438, ggttttttttttttttttttattttt, ggtttttttttttttttttttttttt,
|
||||
27 0 0 0 0 0 1 32 + seq1 36 9 36 chrI 15080483 1302536 1302595 2 21,6, 9,30, 1302536,1302589, tggtttttttttttttttttt,attttt, tggtttttttttttttttttt,attttt,
|
||||
|
||||
-----
|
||||
|
||||
**Parameters**
|
||||
|
||||
- *Minimal Identity (-minIdentity)* : In percent, the minimum sequence identity between the query and target alignment. Default is 90.
|
||||
|
||||
- *Minimal Size of Exact Match (-tileSize)* : The size of a match that will trigger an alignment. Default is 11. Usually between 8 and 12. Must be between 6 and 18.
|
||||
|
||||
- *Number of Mismatch in the Word (-oneOff)* : The number of mismatches allowed in the word (tile size) and still triggers an alignment. Default is 0.
|
||||
|
||||
-----
|
||||
|
||||
**Reference**
|
||||
|
||||
BLAT: Kent, W James, BLAT--the BLAST-like alignment tool. (2002) Genome Research:12(4) 656-664.
|
||||
|
||||
**BLAT**: Kent, W James, BLAT--the BLAST-like alignment tool. (2002) Genome Research:12(4) 656-664.
|
||||
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
|
||||
@@ -49,9 +49,9 @@ def __main__():
|
||||
|
||||
# prepare the database
|
||||
db = {}
|
||||
db_file = open(DB_LOC, "r")
|
||||
for i, line in enumerate(db_file):
|
||||
for i, line in enumerate(file(DB_LOC)):
|
||||
line = line.rstrip('\r\n')
|
||||
if line.startswith('#'): continue
|
||||
fields = line.split()
|
||||
db[(fields[0])] = []
|
||||
for j in xrange(1, len(fields)):
|
||||
@@ -61,6 +61,11 @@ def __main__():
|
||||
retcode = subprocess.call('which megablast 2>&1', shell='True')
|
||||
if retcode < 0:
|
||||
stop_err("Cannot locate megablast.")
|
||||
|
||||
try:
|
||||
assert db.has_key(db_build) is True
|
||||
except:
|
||||
stop_err('Cannot locate the target database. Please check your location file.')
|
||||
|
||||
for chunk in db[(db_build)]:
|
||||
megablast_arguments = ["megablast", "-d", chunk, "-i", query_filename]
|
||||
|
||||
@@ -5,9 +5,9 @@
|
||||
<param name="input_query" type="data" format="fasta" label="Query Sequence"/>
|
||||
<param name="source_select" type="select" display="radio" label="Target database">
|
||||
<option value="test">Ecoli_K12</option>
|
||||
<option value="nt">nt (for nucleotides)</option>
|
||||
<option value="nt">nt</option>
|
||||
</param>
|
||||
<param name="word_size" type="select" label="Word size (-W, length of best perfect match)">
|
||||
<param name="word_size" type="select" label="Word size (-W)" help="Size of best perfect match">
|
||||
<option value="28">28</option>
|
||||
<option value="16">16</option>
|
||||
</param>
|
||||
@@ -55,11 +55,7 @@ Other databases and megablast parameters will be available shortly.
|
||||
|
||||
-----
|
||||
|
||||
**Reference**
|
||||
|
||||
**megablast**: Zhang et al. A Greedy Algorithm for Aligning DNA Sequences. 2000. JCB: 203-214.
|
||||
|
||||
Corresponding command line options:
|
||||
**Parameters**
|
||||
|
||||
- **-W**: Word size
|
||||
- **-p**: Identity percentage cut-off
|
||||
@@ -67,5 +63,11 @@ Other databases and megablast parameters will be available shortly.
|
||||
- **-N**: Type of a discontiguous word template
|
||||
- **-F**: Filter query sequence
|
||||
|
||||
-----
|
||||
|
||||
**Reference**
|
||||
|
||||
**megablast**: Zhang et al. A Greedy Algorithm for Aligning DNA Sequences. 2000. JCB: 203-214.
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
|
||||
Reference in New Issue
Block a user