Merge pull request #218 from erasche/hmmer

Hmmer & Stockholm Datatypes
This commit is contained in:
John Chilton
2015-05-07 07:25:08 -05:00
5 changed files with 202 additions and 8 deletions
+7
View File
@@ -271,6 +271,10 @@
<datatype extension="rdf" type="galaxy.datatypes.graph:Rdf" display_in_upload="true"/>
<!-- Excel datatypes -->
<datatype extension="xlsx" type="galaxy.datatypes.binary:Xlsx" display_in_upload="true" />
<!-- MSA Datatypes -->
<datatype extension="hmm2" type="galaxy.datatypes.msa:Hmmer2" display_in_upload="true" />
<datatype extension="hmm3" type="galaxy.datatypes.msa:Hmmer3" display_in_upload="true" />
<datatype extension="stockholm" type="galaxy.datatypes.msa:Stockholm_1_0" display_in_upload="True" />
</registration>
<sniffers>
<!--
@@ -316,6 +320,9 @@
<sniffer type="galaxy.datatypes.text:Ipynb"/>
<sniffer type="galaxy.datatypes.text:Json"/>
<sniffer type="galaxy.datatypes.sequence:RNADotPlotMatrix"/>
<sniffer type="galaxy.datatypes.msa:Hmmer2" />
<sniffer type="galaxy.datatypes.msa:Hmmer3" />
<sniffer type="galaxy.datatypes.msa:Stockholm_1_0" />
<sniffer type="galaxy.datatypes.images:Jpg"/>
<sniffer type="galaxy.datatypes.images:Png"/>
<sniffer type="galaxy.datatypes.images:Tiff"/>
+168
View File
@@ -0,0 +1,168 @@
from galaxy.datatypes.data import Text
from galaxy.datatypes.binary import Binary
from galaxy.datatypes.data import get_file_peek
from galaxy.datatypes.data import nice_size
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.util import generic_util
import os
import logging
log = logging.getLogger(__name__)
class Hmmer( Text ):
file_ext = "hmm"
def set_peek(self, dataset, is_multi_byte=False):
if not dataset.dataset.purged:
dataset.peek = get_file_peek( dataset.file_name, is_multi_byte=is_multi_byte )
dataset.blurb = "HMMER Database"
else:
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disc'
def display_peek(self, dataset):
try:
return dataset.peek
except:
return "HMMER database (%s)" % ( nice_size( dataset.get_size() ) )
class Hmmer2( Hmmer ):
def sniff(self, filename):
"""HMMER2 files start with HMMER2.0
"""
with open(filename, 'r') as handle:
return handle.read(8) == 'HMMER2.0'
return False
class Hmmer3( Hmmer ):
def sniff(self, filename):
"""HMMER3 files start with HMMER3/f
"""
with open(filename, 'r') as handle:
return handle.read(8) == 'HMMER3/f'
return False
class HmmerPress( Binary ):
"""Class for hmmpress database files."""
file_ext = 'hmmpress'
allow_datatype_change = False
composite_type = 'basic'
def set_peek( self, dataset, is_multi_byte=False ):
"""Set the peek and blurb text."""
if not dataset.dataset.purged:
dataset.peek = "HMMER Binary database"
dataset.blurb = "HMMER Binary database"
else:
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def display_peek( self, dataset ):
"""Create HTML content, used for displaying peek."""
try:
return dataset.peek
except:
return "HMMER3 database (multiple files)"
def __init__(self, **kwd):
Binary.__init__(self, **kwd)
# Binary model
self.add_composite_file('model.hmm.h3m', is_binary=True)
# SSI index for binary model
self.add_composite_file('model.hmm.h3i', is_binary=True)
# Profiles (MSV part)
self.add_composite_file('model.hmm.h3f', is_binary=True)
# Profiles (remained)
self.add_composite_file('model.hmm.h3p', is_binary=True)
Binary.register_unsniffable_binary_ext("hmmpress")
class Stockholm_1_0( Text ):
file_ext = "stockholm"
MetadataElement( name="number_of_alignments", default=0, desc="Number of multiple alignments", readonly=True, visible=True, optional=True, no_value=0 )
def set_peek( self, dataset, is_multi_byte=False ):
if not dataset.dataset.purged:
dataset.peek = get_file_peek( dataset.file_name, is_multi_byte=is_multi_byte )
if (dataset.metadata.number_of_models == 1):
dataset.blurb = "1 alignment"
else:
dataset.blurb = "%s alignments" % dataset.metadata.number_of_models
dataset.peek = get_file_peek( dataset.file_name, is_multi_byte=is_multi_byte )
else:
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disc'
def sniff( self, filename ):
if generic_util.count_special_lines('^#[[:space:]+]STOCKHOLM[[:space:]+]1.0', filename) > 0:
return True
else:
return False
def set_meta( self, dataset, **kwd ):
"""
Set the number of models in dataset.
"""
dataset.metadata.number_of_models = generic_util.count_special_lines('^#[[:space:]+]STOCKHOLM[[:space:]+]1.0', dataset.file_name)
def split( cls, input_datasets, subdir_generator_function, split_params):
"""
Split the input files by model records.
"""
if split_params is None:
return None
if len(input_datasets) > 1:
raise Exception("STOCKHOLM-file splitting does not support multiple files")
input_files = [ds.file_name for ds in input_datasets]
chunk_size = None
if split_params['split_mode'] == 'number_of_parts':
raise Exception('Split mode "%s" is currently not implemented for STOCKHOLM-files.' % split_params['split_mode'])
elif split_params['split_mode'] == 'to_size':
chunk_size = int(split_params['split_size'])
else:
raise Exception('Unsupported split mode %s' % split_params['split_mode'])
def _read_stockholm_records( filename ):
lines = []
with open(filename) as handle:
for line in handle:
lines.append( line )
if line.strip() == '//':
yield lines
lines = []
def _write_part_stockholm_file( accumulated_lines ):
part_dir = subdir_generator_function()
part_path = os.path.join( part_dir, os.path.basename( input_files[0] ) )
part_file = open( part_path, 'w' )
part_file.writelines( accumulated_lines )
part_file.close()
try:
stockholm_records = _read_stockholm_records( input_files[0] )
stockholm_lines_accumulated = []
for counter, stockholm_record in enumerate( stockholm_records, start=1):
stockholm_lines_accumulated.extend( stockholm_record )
if counter % chunk_size == 0:
_write_part_stockholm_file( stockholm_lines_accumulated )
stockholm_lines_accumulated = []
if stockholm_lines_accumulated:
_write_part_stockholm_file( stockholm_lines_accumulated )
except Exception, e:
log.error('Unable to split files: %s' % str(e))
raise
split = classmethod(split)
+1
View File
@@ -23,6 +23,7 @@ import assembly
import ngsindex
import graph
import text
import msa
import galaxy.util
from galaxy.util.odict import odict
from display_applications.application import DisplayApplication
+7 -8
View File
@@ -139,7 +139,7 @@ class Obo( Text ):
def sniff( self, filename ):
"""
Try to guess the Obo filetype.
Try to guess the Obo filetype.
It usually starts with a "format-version:" string and has several stanzas which starts with "id:".
"""
stanza = re.compile(r'^\[.*\]$')
@@ -158,7 +158,7 @@ class Obo( Text ):
class Arff( Text ):
"""
An ARFF (Attribute-Relation File Format) file is an ASCII text file that describes a list of instances sharing a set of attributes.
An ARFF (Attribute-Relation File Format) file is an ASCII text file that describes a list of instances sharing a set of attributes.
http://weka.wikispaces.com/ARFF
"""
file_ext = "arff"
@@ -179,7 +179,7 @@ class Arff( Text ):
def sniff( self, filename ):
"""
Try to guess the Arff filetype.
Try to guess the Arff filetype.
It usually starts with a "format-version:" string and has several stanzas which starts with "id:".
"""
with open( filename ) as handle:
@@ -318,7 +318,7 @@ class SnpSiftDbNSFP( Text ):
composite_type = 'auto_primary_file'
allow_datatype_change = False
"""
## The dbNSFP file is a tabular file with 1 header line
## The dbNSFP file is a tabular file with 1 header line
## The first 4 columns are required to be: chrom pos ref alt
## These match columns 1,2,4,5 of the VCF file
## SnpSift requires the file to be block-gzipped and the indexed with samtools tabix
@@ -335,14 +335,14 @@ class SnpSiftDbNSFP( Text ):
def init_meta( self, dataset, copy_from=None ):
Text.init_meta( self, dataset, copy_from=copy_from )
def generate_primary_file( self, dataset = None ):
"""
"""
This is called only at upload to write the html file
cannot rename the datasets here - they come with the default unfortunately
"""
self.regenerate_primary_file( dataset )
def regenerate_primary_file(self,dataset):
"""
cannot do this until we are setting metadata
cannot do this until we are setting metadata
"""
annotations = "dbNSFP Annotations: %s\n" % ','.join(dataset.metadata.annotation)
f = open(dataset.file_name,'a')
@@ -366,7 +366,7 @@ class SnpSiftDbNSFP( Text ):
lines = buf.splitlines()
headers = lines[0].split('\t')
dataset.metadata.annotation = headers[4:]
except Exception,e:
except Exception,e:
log.warn("set_meta fname: %s %s" % (fname,str(e)))
finally:
fh.close()
@@ -375,4 +375,3 @@ class SnpSiftDbNSFP( Text ):
self.regenerate_primary_file(dataset)
except Exception,e:
log.warn("set_meta fname: %s %s" % (dataset.file_name if dataset and dataset.file_name else 'Unkwown',str(e)))
+19
View File
@@ -0,0 +1,19 @@
import subprocess
def count_special_lines( word, filename, invert=False ):
"""
searching for special 'words' using the grep tool
grep is used to speed up the searching and counting
The number of hits is returned.
"""
try:
cmd = ["grep", "-c"]
if invert:
cmd.append('-v')
cmd.extend([word, filename])
out = subprocess.Popen(cmd, stdout=subprocess.PIPE)
return int(out.communicate()[0].split()[0])
except:
pass
return 0