mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
@@ -271,6 +271,10 @@
|
||||
<datatype extension="rdf" type="galaxy.datatypes.graph:Rdf" display_in_upload="true"/>
|
||||
<!-- Excel datatypes -->
|
||||
<datatype extension="xlsx" type="galaxy.datatypes.binary:Xlsx" display_in_upload="true" />
|
||||
<!-- MSA Datatypes -->
|
||||
<datatype extension="hmm2" type="galaxy.datatypes.msa:Hmmer2" display_in_upload="true" />
|
||||
<datatype extension="hmm3" type="galaxy.datatypes.msa:Hmmer3" display_in_upload="true" />
|
||||
<datatype extension="stockholm" type="galaxy.datatypes.msa:Stockholm_1_0" display_in_upload="True" />
|
||||
</registration>
|
||||
<sniffers>
|
||||
<!--
|
||||
@@ -316,6 +320,9 @@
|
||||
<sniffer type="galaxy.datatypes.text:Ipynb"/>
|
||||
<sniffer type="galaxy.datatypes.text:Json"/>
|
||||
<sniffer type="galaxy.datatypes.sequence:RNADotPlotMatrix"/>
|
||||
<sniffer type="galaxy.datatypes.msa:Hmmer2" />
|
||||
<sniffer type="galaxy.datatypes.msa:Hmmer3" />
|
||||
<sniffer type="galaxy.datatypes.msa:Stockholm_1_0" />
|
||||
<sniffer type="galaxy.datatypes.images:Jpg"/>
|
||||
<sniffer type="galaxy.datatypes.images:Png"/>
|
||||
<sniffer type="galaxy.datatypes.images:Tiff"/>
|
||||
|
||||
@@ -0,0 +1,168 @@
|
||||
from galaxy.datatypes.data import Text
|
||||
from galaxy.datatypes.binary import Binary
|
||||
from galaxy.datatypes.data import get_file_peek
|
||||
from galaxy.datatypes.data import nice_size
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.util import generic_util
|
||||
import os
|
||||
|
||||
|
||||
import logging
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Hmmer( Text ):
|
||||
file_ext = "hmm"
|
||||
|
||||
def set_peek(self, dataset, is_multi_byte=False):
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = get_file_peek( dataset.file_name, is_multi_byte=is_multi_byte )
|
||||
dataset.blurb = "HMMER Database"
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disc'
|
||||
|
||||
def display_peek(self, dataset):
|
||||
try:
|
||||
return dataset.peek
|
||||
except:
|
||||
return "HMMER database (%s)" % ( nice_size( dataset.get_size() ) )
|
||||
|
||||
|
||||
class Hmmer2( Hmmer ):
|
||||
|
||||
def sniff(self, filename):
|
||||
"""HMMER2 files start with HMMER2.0
|
||||
"""
|
||||
with open(filename, 'r') as handle:
|
||||
return handle.read(8) == 'HMMER2.0'
|
||||
return False
|
||||
|
||||
|
||||
class Hmmer3( Hmmer ):
|
||||
|
||||
def sniff(self, filename):
|
||||
"""HMMER3 files start with HMMER3/f
|
||||
"""
|
||||
with open(filename, 'r') as handle:
|
||||
return handle.read(8) == 'HMMER3/f'
|
||||
return False
|
||||
|
||||
|
||||
class HmmerPress( Binary ):
|
||||
"""Class for hmmpress database files."""
|
||||
file_ext = 'hmmpress'
|
||||
allow_datatype_change = False
|
||||
composite_type = 'basic'
|
||||
|
||||
def set_peek( self, dataset, is_multi_byte=False ):
|
||||
"""Set the peek and blurb text."""
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = "HMMER Binary database"
|
||||
dataset.blurb = "HMMER Binary database"
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def display_peek( self, dataset ):
|
||||
"""Create HTML content, used for displaying peek."""
|
||||
try:
|
||||
return dataset.peek
|
||||
except:
|
||||
return "HMMER3 database (multiple files)"
|
||||
|
||||
def __init__(self, **kwd):
|
||||
Binary.__init__(self, **kwd)
|
||||
# Binary model
|
||||
self.add_composite_file('model.hmm.h3m', is_binary=True)
|
||||
# SSI index for binary model
|
||||
self.add_composite_file('model.hmm.h3i', is_binary=True)
|
||||
# Profiles (MSV part)
|
||||
self.add_composite_file('model.hmm.h3f', is_binary=True)
|
||||
# Profiles (remained)
|
||||
self.add_composite_file('model.hmm.h3p', is_binary=True)
|
||||
|
||||
Binary.register_unsniffable_binary_ext("hmmpress")
|
||||
|
||||
|
||||
class Stockholm_1_0( Text ):
|
||||
file_ext = "stockholm"
|
||||
|
||||
MetadataElement( name="number_of_alignments", default=0, desc="Number of multiple alignments", readonly=True, visible=True, optional=True, no_value=0 )
|
||||
|
||||
def set_peek( self, dataset, is_multi_byte=False ):
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = get_file_peek( dataset.file_name, is_multi_byte=is_multi_byte )
|
||||
if (dataset.metadata.number_of_models == 1):
|
||||
dataset.blurb = "1 alignment"
|
||||
else:
|
||||
dataset.blurb = "%s alignments" % dataset.metadata.number_of_models
|
||||
dataset.peek = get_file_peek( dataset.file_name, is_multi_byte=is_multi_byte )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disc'
|
||||
|
||||
def sniff( self, filename ):
|
||||
if generic_util.count_special_lines('^#[[:space:]+]STOCKHOLM[[:space:]+]1.0', filename) > 0:
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
|
||||
def set_meta( self, dataset, **kwd ):
|
||||
"""
|
||||
|
||||
Set the number of models in dataset.
|
||||
"""
|
||||
dataset.metadata.number_of_models = generic_util.count_special_lines('^#[[:space:]+]STOCKHOLM[[:space:]+]1.0', dataset.file_name)
|
||||
|
||||
def split( cls, input_datasets, subdir_generator_function, split_params):
|
||||
"""
|
||||
|
||||
Split the input files by model records.
|
||||
"""
|
||||
if split_params is None:
|
||||
return None
|
||||
|
||||
if len(input_datasets) > 1:
|
||||
raise Exception("STOCKHOLM-file splitting does not support multiple files")
|
||||
input_files = [ds.file_name for ds in input_datasets]
|
||||
|
||||
chunk_size = None
|
||||
if split_params['split_mode'] == 'number_of_parts':
|
||||
raise Exception('Split mode "%s" is currently not implemented for STOCKHOLM-files.' % split_params['split_mode'])
|
||||
elif split_params['split_mode'] == 'to_size':
|
||||
chunk_size = int(split_params['split_size'])
|
||||
else:
|
||||
raise Exception('Unsupported split mode %s' % split_params['split_mode'])
|
||||
|
||||
def _read_stockholm_records( filename ):
|
||||
lines = []
|
||||
with open(filename) as handle:
|
||||
for line in handle:
|
||||
lines.append( line )
|
||||
if line.strip() == '//':
|
||||
yield lines
|
||||
lines = []
|
||||
|
||||
def _write_part_stockholm_file( accumulated_lines ):
|
||||
part_dir = subdir_generator_function()
|
||||
part_path = os.path.join( part_dir, os.path.basename( input_files[0] ) )
|
||||
part_file = open( part_path, 'w' )
|
||||
part_file.writelines( accumulated_lines )
|
||||
part_file.close()
|
||||
|
||||
try:
|
||||
|
||||
stockholm_records = _read_stockholm_records( input_files[0] )
|
||||
stockholm_lines_accumulated = []
|
||||
for counter, stockholm_record in enumerate( stockholm_records, start=1):
|
||||
stockholm_lines_accumulated.extend( stockholm_record )
|
||||
if counter % chunk_size == 0:
|
||||
_write_part_stockholm_file( stockholm_lines_accumulated )
|
||||
stockholm_lines_accumulated = []
|
||||
if stockholm_lines_accumulated:
|
||||
_write_part_stockholm_file( stockholm_lines_accumulated )
|
||||
except Exception, e:
|
||||
log.error('Unable to split files: %s' % str(e))
|
||||
raise
|
||||
split = classmethod(split)
|
||||
@@ -23,6 +23,7 @@ import assembly
|
||||
import ngsindex
|
||||
import graph
|
||||
import text
|
||||
import msa
|
||||
import galaxy.util
|
||||
from galaxy.util.odict import odict
|
||||
from display_applications.application import DisplayApplication
|
||||
|
||||
@@ -139,7 +139,7 @@ class Obo( Text ):
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Try to guess the Obo filetype.
|
||||
Try to guess the Obo filetype.
|
||||
It usually starts with a "format-version:" string and has several stanzas which starts with "id:".
|
||||
"""
|
||||
stanza = re.compile(r'^\[.*\]$')
|
||||
@@ -158,7 +158,7 @@ class Obo( Text ):
|
||||
|
||||
class Arff( Text ):
|
||||
"""
|
||||
An ARFF (Attribute-Relation File Format) file is an ASCII text file that describes a list of instances sharing a set of attributes.
|
||||
An ARFF (Attribute-Relation File Format) file is an ASCII text file that describes a list of instances sharing a set of attributes.
|
||||
http://weka.wikispaces.com/ARFF
|
||||
"""
|
||||
file_ext = "arff"
|
||||
@@ -179,7 +179,7 @@ class Arff( Text ):
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Try to guess the Arff filetype.
|
||||
Try to guess the Arff filetype.
|
||||
It usually starts with a "format-version:" string and has several stanzas which starts with "id:".
|
||||
"""
|
||||
with open( filename ) as handle:
|
||||
@@ -318,7 +318,7 @@ class SnpSiftDbNSFP( Text ):
|
||||
composite_type = 'auto_primary_file'
|
||||
allow_datatype_change = False
|
||||
"""
|
||||
## The dbNSFP file is a tabular file with 1 header line
|
||||
## The dbNSFP file is a tabular file with 1 header line
|
||||
## The first 4 columns are required to be: chrom pos ref alt
|
||||
## These match columns 1,2,4,5 of the VCF file
|
||||
## SnpSift requires the file to be block-gzipped and the indexed with samtools tabix
|
||||
@@ -335,14 +335,14 @@ class SnpSiftDbNSFP( Text ):
|
||||
def init_meta( self, dataset, copy_from=None ):
|
||||
Text.init_meta( self, dataset, copy_from=copy_from )
|
||||
def generate_primary_file( self, dataset = None ):
|
||||
"""
|
||||
"""
|
||||
This is called only at upload to write the html file
|
||||
cannot rename the datasets here - they come with the default unfortunately
|
||||
"""
|
||||
self.regenerate_primary_file( dataset )
|
||||
def regenerate_primary_file(self,dataset):
|
||||
"""
|
||||
cannot do this until we are setting metadata
|
||||
cannot do this until we are setting metadata
|
||||
"""
|
||||
annotations = "dbNSFP Annotations: %s\n" % ','.join(dataset.metadata.annotation)
|
||||
f = open(dataset.file_name,'a')
|
||||
@@ -366,7 +366,7 @@ class SnpSiftDbNSFP( Text ):
|
||||
lines = buf.splitlines()
|
||||
headers = lines[0].split('\t')
|
||||
dataset.metadata.annotation = headers[4:]
|
||||
except Exception,e:
|
||||
except Exception,e:
|
||||
log.warn("set_meta fname: %s %s" % (fname,str(e)))
|
||||
finally:
|
||||
fh.close()
|
||||
@@ -375,4 +375,3 @@ class SnpSiftDbNSFP( Text ):
|
||||
self.regenerate_primary_file(dataset)
|
||||
except Exception,e:
|
||||
log.warn("set_meta fname: %s %s" % (dataset.file_name if dataset and dataset.file_name else 'Unkwown',str(e)))
|
||||
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
import subprocess
|
||||
|
||||
|
||||
def count_special_lines( word, filename, invert=False ):
|
||||
"""
|
||||
searching for special 'words' using the grep tool
|
||||
grep is used to speed up the searching and counting
|
||||
The number of hits is returned.
|
||||
"""
|
||||
try:
|
||||
cmd = ["grep", "-c"]
|
||||
if invert:
|
||||
cmd.append('-v')
|
||||
cmd.extend([word, filename])
|
||||
out = subprocess.Popen(cmd, stdout=subprocess.PIPE)
|
||||
return int(out.communicate()[0].split()[0])
|
||||
except:
|
||||
pass
|
||||
return 0
|
||||
Reference in New Issue
Block a user