Use fasta HDAs to specify custom builds. Do automatic conversions from fasta to 2bit and len; also, do conversion from len to linecount so that chrom/contig count is available. Finally, refactor 'compute fasta length' code out of tools and into converters.

This commit is contained in:
Jeremy Goecks
2011-09-20 11:19:44 -04:00
parent a4a858c40d
commit a7c2899448
10 changed files with 167 additions and 81 deletions
+1
View File
@@ -135,6 +135,7 @@
<datatype extension="tabular" type="galaxy.datatypes.tabular:Tabular" display_in_upload="true"/>
<datatype extension="twobit" type="galaxy.datatypes.binary:TwoBit" mimetype="application/octet-stream" display_in_upload="true"/>
<datatype extension="txt" type="galaxy.datatypes.data:Text" display_in_upload="true"/>
<datatype extension="linecount" type="galaxy.datatypes.data:LineCount" display_in_upload="false"/>
<datatype extension="memexml" type="galaxy.datatypes.xml:MEMEXml" mimetype="application/xml" display_in_upload="true"/>
<datatype extension="cisml" type="galaxy.datatypes.xml:CisML" mimetype="application/xml" display_in_upload="true"/>
<datatype extension="blastxml" type="galaxy.datatypes.xml:BlastXml" mimetype="application/xml" display_in_upload="true"/>
@@ -0,0 +1,13 @@
<tool id="CONVERTER_fasta_to_2bit" name="Convert FASTA to 2bit" version="1.0.0">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<command>faToTwoBit $input $output</command>
<inputs>
<param name="input" type="data" format="fasta" label="Fasta file"/>
</inputs>
<outputs>
<data name="output" format="twobit"/>
</outputs>
<help>
</help>
</tool>
@@ -0,0 +1,48 @@
#!/usr/bin/env python
"""
Input: fasta, int
Output: tabular
Return titles with lengths of corresponding seq
"""
import sys, os
assert sys.version_info[:2] >= ( 2, 4 )
def compute_fasta_length( fasta_file, out_file, keep_first_char ):
infile = fasta_file
out = open( out_file, 'w')
keep_first_char = int( keep_first_char )
fasta_title = ''
seq_len = 0
# number of char to keep in the title
if keep_first_char == 0:
keep_first_char = None
else:
keep_first_char += 1
first_entry = True
for line in open( infile ):
line = line.strip()
if not line or line.startswith( '#' ):
continue
if line[0] == '>':
if first_entry == False:
out.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first_char ], seq_len ) )
else:
first_entry = False
fasta_title = line
seq_len = 0
else:
seq_len += len(line)
# last fasta-entry
out.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first_char ], seq_len ) )
out.close()
if __name__ == "__main__" :
compute_fasta_length( sys.argv[1], sys.argv[2], sys.argv[3] )
@@ -0,0 +1,13 @@
<tool id="CONVERTER_fasta_to_len" name="Convert FASTA to len file" version="1.0.0">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<command interpreter="python">fasta_to_len.py $input $output 0</command>
<inputs>
<param name="input" type="data" format="fasta" label="Fasta file"/>
</inputs>
<outputs>
<data name="output" format="len"/>
</outputs>
<help>
</help>
</tool>
@@ -0,0 +1,13 @@
<tool id="CONVERTER_len_to_linecount" name="Convert Len file to Linecount" version="1.0.0">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<command>wc -l $input | awk '{print $1}' &gt; $output </command>
<inputs>
<param name="input" type="data" format="len" label="Fasta file"/>
</inputs>
<outputs>
<data name="output" format="linecount"/>
</outputs>
<help>
</help>
</tool>
+7
View File
@@ -446,6 +446,13 @@ class Text( Data ):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
class LineCount( Text ):
"""
Dataset contains a single line with a single integer that denotes the
line count for a related dataset. Used for custom builds.
"""
pass
class Newick( Text ):
pass
+60 -35
View File
@@ -1165,6 +1165,9 @@ class User( BaseController, UsesFormDefinitions ):
@web.expose
@web.require_login()
def dbkeys( self, trans, **kwds ):
#
# Process arguments and add/delete build.
#
user = trans.user
message = None
lines_skipped = 0
@@ -1178,57 +1181,79 @@ class User( BaseController, UsesFormDefinitions ):
else:
dbkeys = from_json_string(user.preferences['dbkeys'])
if 'delete' in kwds:
# Delete a build.
key = kwds.get('key', '')
if key and key in dbkeys:
del dbkeys[key]
elif 'add' in kwds:
name = kwds.get('name', '')
key = kwds.get('key', '')
len_file = kwds.get('len_file', None)
if getattr(len_file, "file", None): # Check if it's a FieldStorage object
len_text = len_file.file.read()
else:
len_text = kwds.get('len_text', '')
if not name or not key or not len_text:
# Add new custom build.
name = kwds.get('name', '')
key = kwds.get('key', '')
dataset_id = kwds.get('dataset_id', '')
if not name or not key or not dataset_id:
message = "You must specify values for all the fields."
elif key in dbkeys:
message = "There is already a custom build with that key. Delete it first if you want to replace it."
else:
# Create new len file
new_len = trans.app.model.HistoryDatasetAssociation( extension="len", create_dataset=True, sa_session=trans.sa_session )
trans.sa_session.add( new_len )
new_len.name = name
new_len.visible = False
new_len.state = trans.app.model.Job.states.OK
new_len.info = "custom build .len file"
trans.sa_session.flush()
counter = 0
f = open(new_len.file_name, "w")
# LEN files have format:
# <chrom_name><tab><chrom_length>
for line in len_text.split("\n"):
lst = line.strip().rsplit(None, 1) # Splits at the last whitespace in the line
if not lst or len(lst) < 2:
lines_skipped += 1
continue
chrom, length = lst[0], lst[1]
try:
length = int(length)
except ValueError:
lines_skipped += 1
continue
counter += 1
f.write("%s\t%s\n" % (chrom, length))
f.close()
dbkeys[key] = { "name": name, "len": new_len.id, "count": counter }
dataset_id = trans.security.decode_id( dataset_id )
dbkeys[key] = { "name": name, "fasta": dataset_id }
# Save builds.
# TODO: use database table to save builds.
user.preferences['dbkeys'] = to_json_string(dbkeys)
trans.sa_session.flush()
#
# Display custom builds page.
#
# Add chrom/contig count to dbkeys dict.
updated = False
for key, attributes in dbkeys.items():
if 'count' in attributes:
# Already have count, so do nothing.
continue
# Get len file.
fasta_dataset = trans.app.model.HistoryDatasetAssociation.get( attributes[ 'fasta' ] )
len_dataset = fasta_dataset.get_converted_dataset( trans, "len" )
# HACK: need to request dataset again b/c get_converted_dataset()
# doesn't return dataset (as it probably should).
len_dataset = fasta_dataset.get_converted_dataset( trans, "len" )
if len_dataset.state == trans.app.model.Job.states.ERROR:
# Can't use len dataset.
continue
# Get chrom count file.
# NOTE: this conversion doesn't work well with set_metadata_externally=False
# because the conversion occurs before metadata can be set; the
# dataset is marked as deleted and a subsequent conversion is run.
chrom_count_dataset = len_dataset.get_converted_dataset( trans, "linecount" )
if not chrom_count_dataset or chrom_count_dataset.state != trans.app.model.Job.states.OK:
# No valid linecount dataset.
continue
else:
# Set chrom count.
chrom_count = int( open( chrom_count_dataset.file_name ).readline() )
attributes[ 'count' ] = chrom_count
updated = True
if updated:
user.preferences['dbkeys'] = to_json_string(dbkeys)
trans.sa_session.flush()
# Potential genome data for custom builds is limited to fasta datasets in current history for now.
fasta_hdas = trans.sa_session.query( model.HistoryDatasetAssociation ) \
.filter_by( history=trans.history, extension="fasta", deleted=False ) \
.order_by( model.HistoryDatasetAssociation.hid.desc() )
return trans.fill_template( 'user/dbkeys.mako',
user=user,
dbkeys=dbkeys,
message=message,
installed_len_files=self.installed_len_files,
lines_skipped=lines_skipped,
fasta_hdas=fasta_hdas,
use_panels=kwds.get( 'use_panels', None ) )
@web.expose
@web.require_login()
+8 -4
View File
@@ -103,6 +103,8 @@
## </pre>
% if 'count' in dct:
${dct['count']}
% else:
working
% endif
</td>
<td><form action="dbkeys" method="post"><input type="hidden" name="key" value="${key}" /><input type="submit" name="delete" value="Delete" /></form></td>
@@ -133,10 +135,12 @@
<input type="text" id="key" name="key" />
</div>
<div class="form-row">
<label for="len_file">Upload .len file:</label>
<input type="file" id="len_file" name="len_file" /><br />
<label for="len_text">Or enter/paste length info below:</label>
<textarea id="len_text" name="len_text" cols="30" rows="8"></textarea>
<label for="len_file">Build Genome:</label>
<select name="dataset_id">
%for dataset in fasta_hdas:
<option value="${trans.security.encode_id( dataset.id )}">${dataset.hid}: ${dataset.name}</option>
%endfor
</select>
</div>
<div class="form-row"><input type="submit" name="add" value="Submit"/></div>
+4 -42
View File
@@ -1,47 +1,9 @@
#!/usr/bin/env python
"""
Input: fasta, int
Output: tabular
Return titles with lengths of corresponding seq
Uses fasta_to_len converter code.
"""
import sys, os
import sys
from galaxy.datatypes.converters.fasta_to_len import compute_fasta_length
assert sys.version_info[:2] >= ( 2, 4 )
def __main__():
infile = sys.argv[1]
out = open( sys.argv[2], 'w')
keep_first_char = int( sys.argv[3] )
fasta_title = ''
seq_len = 0
# number of char to keep in the title
if keep_first_char == 0:
keep_first_char = None
else:
keep_first_char += 1
first_entry = True
for line in open( infile ):
line = line.strip()
if not line or line.startswith( '#' ):
continue
if line[0] == '>':
if first_entry == False:
out.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first_char ], seq_len ) )
else:
first_entry = False
fasta_title = line
seq_len = 0
else:
seq_len += len(line)
# last fasta-entry
out.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first_char ], seq_len ) )
out.close()
if __name__ == "__main__" : __main__()
compute_fasta_length( sys.argv[1], sys.argv[2], sys.argv[3])