Move idpDB and MzSQlite formats from proteomics.py to binary.py, because the SQLite sniffer lives there, and more specialized formats must be registered for sniffing before the generic format

Add proper idpDB sniffing logic (akin to MzSQlite) and test case
Add H5 sniffing based on magic number and test case
Add SQLite/IdpDB/MzSQlite/GeminiSQLite and H5 to default datatypes in registry.py
This commit is contained in:
chambm
2015-12-02 18:27:33 -06:00
parent 9e28f2e786
commit 97415de3ce
7 changed files with 134 additions and 40 deletions
+6 -4
View File
@@ -107,7 +107,7 @@
<display file="igb/gtf.xml" />
</datatype>
<datatype extension="toolshed.gz" type="galaxy.datatypes.binary:Binary" mimetype="multipart/x-gzip" subclass="True" />
<datatype extension="h5" type="galaxy.datatypes.binary:Binary" mimetype="application/octet-stream" subclass="True" display_in_upload="True"/>
<datatype extension="h5" type="galaxy.datatypes.binary:H5" mimetype="application/octet-stream" display_in_upload="True"/>
<datatype extension="html" type="galaxy.datatypes.images:Html" mimetype="text/html"/>
<datatype extension="interval" type="galaxy.datatypes.interval:Interval" display_in_upload="true" description="File must start with definition line in the following format (columns may be in any order)." >
<converter file="interval_to_bed_converter.xml" target_datatype="bed"/>
@@ -168,7 +168,7 @@
<datatype extension="mzxml" type="galaxy.datatypes.proteomics:MzXML" mimetype="application/xml" display_in_upload="true" />
<datatype extension="ms2" type="galaxy.datatypes.proteomics:Ms2" display_in_upload="true" />
<datatype extension="mzq" type="galaxy.datatypes.proteomics:MzQuantML" mimetype="application/xml" display_in_upload="true" />
<datatype extension="mz.sqlite" type="galaxy.datatypes.proteomics:MzSQlite" mimetype="application/octet-stream" display_in_upload="true" />
<datatype extension="mz.sqlite" type="galaxy.datatypes.binary:MzSQlite" mimetype="application/octet-stream" display_in_upload="true" />
<datatype extension="traml" type="galaxy.datatypes.proteomics:TraML" mimetype="application/xml" display_in_upload="true" />
<datatype extension="featurexml" type="galaxy.datatypes.proteomics:FeatureXML" mimetype="application/xml" display_in_upload="true" />
<datatype extension="consensusxml" type="galaxy.datatypes.proteomics:ConsensusXML" mimetype="application/xml" display_in_upload="true" />
@@ -177,7 +177,7 @@
<datatype extension="splib_noindex" type="galaxy.datatypes.proteomics:SPLibNoIndex" display_in_upload="true" />
<datatype extension="splib" type="galaxy.datatypes.proteomics:SPLib" display_in_upload="true" />
<datatype extension="hlf" type="galaxy.datatypes.proteomics:XHunterAslFormat" mimetype="application/octet-stream" display_in_upload="true" />
<datatype extension="idpdb" type="galaxy.datatypes.proteomics:IdpDB" display_in_upload="true" />
<datatype extension="idpdb" type="galaxy.datatypes.binary:IdpDB" mimetype="application/octet-stream" display_in_upload="true" />
<datatype extension="sf3" type="galaxy.datatypes.proteomics:Sf3" display_in_upload="true" />
<datatype extension="cps" type="galaxy.datatypes.binary:Binary" subclass="True" display_in_upload="true" />
<datatype extension="ct" type="galaxy.datatypes.tabular:ConnectivityTable" display_in_upload="True"/>
@@ -446,8 +446,10 @@
<sniffer type="galaxy.datatypes.tabular:Vcf"/>
<sniffer type="galaxy.datatypes.binary:TwoBit"/>
<sniffer type="galaxy.datatypes.binary:GeminiSQLite"/>
<sniffer type="galaxy.datatypes.proteomics:MzSQlite"/>
<sniffer type="galaxy.datatypes.binary:MzSQlite"/>
<sniffer type="galaxy.datatypes.binary:IdpDB"/>
<sniffer type="galaxy.datatypes.binary:SQlite"/>
<sniffer type="galaxy.datatypes.binary:H5"/>
<sniffer type="galaxy.datatypes.binary:Bam"/>
<sniffer type="galaxy.datatypes.binary:CRAM"/>
<sniffer type="galaxy.datatypes.binary:Sff"/>
+105 -5
View File
@@ -540,12 +540,36 @@ Binary.register_sniffable_binary_format("bcf", "bcf", Bcf)
class H5( Binary ):
"""Class describing an HDF5 file"""
"""
Class describing an HDF5 file
>>> from galaxy.datatypes.sniff import get_test_fname
>>> fname = get_test_fname( 'test.mz5' )
>>> H5().sniff( fname )
True
>>> fname = get_test_fname( 'interval.interval' )
>>> H5().sniff( fname )
False
"""
file_ext = "h5"
def __init__( self, **kwd ):
Binary.__init__( self, **kwd )
self._magic = binascii.unhexlify("894844460d0a1a0a")
def sniff( self, filename ):
# The first 8 bytes of any hdf5 file are 0x894844460d0a1a0a
try:
header = open( filename ).read(8)
if header == self._magic:
return True
return False
except:
return False
def set_peek( self, dataset, is_multi_byte=False ):
if not dataset.dataset.purged:
dataset.peek = "Binary h5 file"
dataset.peek = "Binary HDF5 file"
dataset.blurb = nice_size( dataset.get_size() )
else:
dataset.peek = 'file does not exist'
@@ -555,9 +579,9 @@ class H5( Binary ):
try:
return dataset.peek
except:
return "Binary h5 sequence file (%s)" % ( nice_size( dataset.get_size() ) )
return "Binary HDF5 file (%s)" % ( nice_size( dataset.get_size() ) )
Binary.register_unsniffable_binary_ext("h5")
Binary.register_sniffable_binary_format("h5", "h5", H5)
class Scf( Binary ):
@@ -847,8 +871,84 @@ class GeminiSQLite( SQlite ):
except:
return "Gemini SQLite Database, version %s" % ( dataset.metadata.gemini_version or 'unknown' )
class MzSQlite( SQlite ):
"""Class describing a Proteomics Sqlite database """
file_ext = "mz.sqlite"
def set_meta( self, dataset, overwrite=True, **kwd ):
super( MzSQlite, self ).set_meta( dataset, overwrite=overwrite, **kwd )
def sniff( self, filename ):
if super( MzSQlite, self ).sniff( filename ):
mz_table_names = ["DBSequence", "Modification", "Peaks", "Peptide", "PeptideEvidence", "Score", "SearchDatabase", "Source", "SpectraData", "Spectrum", "SpectrumIdentification"]
try:
conn = sqlite.connect( filename )
c = conn.cursor()
tables_query = "SELECT name FROM sqlite_master WHERE type='table' ORDER BY name"
result = c.execute( tables_query ).fetchall()
result = map( lambda x: x[0], result )
for table_name in mz_table_names:
if table_name not in result:
return False
return True
except Exception, e:
log.warn( '%s, sniff Exception: %s', self, e )
return False
class IdpDB( SQlite ):
"""
Class describing an IDPicker 3 idpDB (sqlite) database
>>> from galaxy.datatypes.sniff import get_test_fname
>>> fname = get_test_fname( 'test.idpDB' )
>>> IdpDB().sniff( fname )
True
>>> fname = get_test_fname( 'interval.interval' )
>>> IdpDB().sniff( fname )
False
"""
file_ext = "idpdb"
def set_meta( self, dataset, overwrite=True, **kwd ):
super( IdpDB, self ).set_meta( dataset, overwrite=overwrite, **kwd )
def sniff( self, filename ):
if super( IdpDB, self ).sniff( filename ):
mz_table_names = ["About", "Analysis", "AnalysisParameter", "PeptideSpectrumMatch", "Spectrum", "SpectrumSource"]
try:
conn = sqlite.connect( filename )
c = conn.cursor()
tables_query = "SELECT name FROM sqlite_master WHERE type='table' ORDER BY name"
result = c.execute( tables_query ).fetchall()
result = map( lambda x: x[0], result )
for table_name in mz_table_names:
if table_name not in result:
return False
return True
except Exception, e:
log.warn( '%s, sniff Exception: %s', self, e )
return False
def set_peek( self, dataset, is_multi_byte=False ):
if not dataset.dataset.purged:
dataset.peek = "IDPickerDB SQLite file"
dataset.blurb = nice_size( dataset.get_size() )
else:
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def display_peek( self, dataset ):
try:
return dataset.peek
except:
return "IDPickerDB SQLite file (%s)" % ( nice_size( dataset.get_size() ) )
Binary.register_sniffable_binary_format( "gemini.sqlite", "gemini.sqlite", GeminiSQLite )
# FIXME: We need to register gemini.sqlite before sqlite, since register_sniffable_binary_format and is_sniffable_binary called in upload.py
Binary.register_sniffable_binary_format( "idpdb", "idpdb", IdpDB )
Binary.register_sniffable_binary_format( "mz.sqlite", "mz.sqlite", MzSQlite )
# FIXME: We need to register specialized sqlite formats before sqlite, since register_sniffable_binary_format and is_sniffable_binary called in upload.py
# ignores sniff order declared in datatypes_conf.xml
Binary.register_sniffable_binary_format("sqlite", "sqlite", SQlite)
+2 -31
View File
@@ -6,11 +6,11 @@ import logging
import re
from galaxy.datatypes import data
from galaxy.datatypes.binary import Binary, SQlite
from galaxy.datatypes.binary import Binary
from galaxy.datatypes.data import Text
from galaxy.datatypes.tabular import Tabular
from galaxy.datatypes.xml import GenericXml
from galaxy.util import nice_size, sqlite
from galaxy.util import nice_size
log = logging.getLogger(__name__)
@@ -53,10 +53,6 @@ class Wiff(Binary):
Binary.register_sniffable_binary_format("wiff", "wiff", Wiff )
class IdpDB(Binary):
file_ext = "idpDB"
class PepXmlReport(Tabular):
"""pepxml converted to tabular report"""
file_ext = "tsv"
@@ -406,28 +402,3 @@ class XHunterAslFormat(Binary):
class Sf3(Binary):
"""Class describing a Scaffold SF3 files"""
file_ext = "sf3"
class MzSQlite( SQlite ):
"""Class describing a Proteomics Sqlite database """
file_ext = "mz.sqlite"
def set_meta( self, dataset, overwrite=True, **kwd ):
super( MzSQlite, self ).set_meta( dataset, overwrite=overwrite, **kwd )
def sniff( self, filename ):
if super( MzSQlite, self ).sniff( filename ):
mz_table_names = ["DBSequence", "Modification", "Peaks", "Peptide", "PeptideEvidence", "Score", "SearchDatabase", "Source", "SpectraData", "Spectrum", "SpectrumIdentification"]
try:
conn = sqlite.connect( filename )
c = conn.cursor()
tables_query = "SELECT name FROM sqlite_master WHERE type='table' ORDER BY name"
result = c.execute( tables_query ).fetchall()
result = map( lambda x: x[0], result )
for table_name in mz_table_names:
if table_name not in result:
return False
return True
except Exception, e:
log.warn( '%s, sniff Exception: %s', self, e )
return False
+15
View File
@@ -651,18 +651,23 @@ class Registry( object ):
'coverage' : coverage.LastzCoverage(),
'customtrack' : interval.CustomTrack(),
'csfasta' : sequence.csFasta(),
'db3' : binary.SQlite(),
'fasta' : sequence.Fasta(),
'eland' : tabular.Eland(),
'fastq' : sequence.Fastq(),
'fastqsanger' : sequence.FastqSanger(),
'gemini.sqlite' : binary.GeminiSQLite(),
'gtf' : interval.Gtf(),
'gff' : interval.Gff(),
'gff3' : interval.Gff3(),
'genetrack' : tracks.GeneTrack(),
'h5' : binary.H5(),
'idpdb' : binary.IdpDB(),
'interval' : interval.Interval(),
'laj' : images.Laj(),
'lav' : sequence.Lav(),
'maf' : sequence.Maf(),
'mz.sqlite' : binary.MzSQlite(),
'pileup' : tabular.Pileup(),
'qualsolid' : qualityscore.QualityScoreSOLiD(),
'qualsolexa' : qualityscore.QualityScoreSolexa(),
@@ -684,18 +689,23 @@ class Registry( object ):
'bed' : 'text/plain',
'customtrack' : 'text/plain',
'csfasta' : 'text/plain',
'db3' : 'application/octet-stream',
'eland' : 'application/octet-stream',
'fasta' : 'text/plain',
'fastq' : 'text/plain',
'fastqsanger' : 'text/plain',
'gemini.sqlite' : 'application/octet-stream',
'gtf' : 'text/plain',
'gff' : 'text/plain',
'gff3' : 'text/plain',
'h5' : 'application/octet-stream',
'idpdb' : 'application/octet-stream',
'interval' : 'text/plain',
'laj' : 'text/plain',
'lav' : 'text/plain',
'maf' : 'text/plain',
'memexml' : 'application/xml',
'mz.sqlite' : 'application/octet-stream',
'pileup' : 'text/plain',
'qualsolid' : 'text/plain',
'qualsolexa' : 'text/plain',
@@ -720,6 +730,11 @@ class Registry( object ):
self.sniff_order = [
binary.Bam(),
binary.Sff(),
binary.H5(),
binary.GeminiSQLite(),
binary.MzSQlite(),
binary.IdpDB(),
binary.SQlite(),
xml.GenericXml(),
sequence.Maf(),
sequence.Lav(),
+6
View File
@@ -310,6 +310,12 @@ def guess_ext( fname, sniff_order=None, is_multi_byte=False ):
>>> fname = get_test_fname('3unsorted.bam')
>>> guess_ext(fname)
'bam'
>>> fname = get_test_fname('test.idpDB')
>>> guess_ext(fname)
'idpdb'
>>> fname = get_test_fname('test.mz5')
>>> guess_ext(fname)
'h5'
"""
if sniff_order is None:
datatypes_registry = registry.Registry()
Binary file not shown.
Binary file not shown.