mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Move idpDB and MzSQlite formats from proteomics.py to binary.py, because the SQLite sniffer lives there, and more specialized formats must be registered for sniffing before the generic format
Add proper idpDB sniffing logic (akin to MzSQlite) and test case Add H5 sniffing based on magic number and test case Add SQLite/IdpDB/MzSQlite/GeminiSQLite and H5 to default datatypes in registry.py
This commit is contained in:
@@ -107,7 +107,7 @@
|
||||
<display file="igb/gtf.xml" />
|
||||
</datatype>
|
||||
<datatype extension="toolshed.gz" type="galaxy.datatypes.binary:Binary" mimetype="multipart/x-gzip" subclass="True" />
|
||||
<datatype extension="h5" type="galaxy.datatypes.binary:Binary" mimetype="application/octet-stream" subclass="True" display_in_upload="True"/>
|
||||
<datatype extension="h5" type="galaxy.datatypes.binary:H5" mimetype="application/octet-stream" display_in_upload="True"/>
|
||||
<datatype extension="html" type="galaxy.datatypes.images:Html" mimetype="text/html"/>
|
||||
<datatype extension="interval" type="galaxy.datatypes.interval:Interval" display_in_upload="true" description="File must start with definition line in the following format (columns may be in any order)." >
|
||||
<converter file="interval_to_bed_converter.xml" target_datatype="bed"/>
|
||||
@@ -168,7 +168,7 @@
|
||||
<datatype extension="mzxml" type="galaxy.datatypes.proteomics:MzXML" mimetype="application/xml" display_in_upload="true" />
|
||||
<datatype extension="ms2" type="galaxy.datatypes.proteomics:Ms2" display_in_upload="true" />
|
||||
<datatype extension="mzq" type="galaxy.datatypes.proteomics:MzQuantML" mimetype="application/xml" display_in_upload="true" />
|
||||
<datatype extension="mz.sqlite" type="galaxy.datatypes.proteomics:MzSQlite" mimetype="application/octet-stream" display_in_upload="true" />
|
||||
<datatype extension="mz.sqlite" type="galaxy.datatypes.binary:MzSQlite" mimetype="application/octet-stream" display_in_upload="true" />
|
||||
<datatype extension="traml" type="galaxy.datatypes.proteomics:TraML" mimetype="application/xml" display_in_upload="true" />
|
||||
<datatype extension="featurexml" type="galaxy.datatypes.proteomics:FeatureXML" mimetype="application/xml" display_in_upload="true" />
|
||||
<datatype extension="consensusxml" type="galaxy.datatypes.proteomics:ConsensusXML" mimetype="application/xml" display_in_upload="true" />
|
||||
@@ -177,7 +177,7 @@
|
||||
<datatype extension="splib_noindex" type="galaxy.datatypes.proteomics:SPLibNoIndex" display_in_upload="true" />
|
||||
<datatype extension="splib" type="galaxy.datatypes.proteomics:SPLib" display_in_upload="true" />
|
||||
<datatype extension="hlf" type="galaxy.datatypes.proteomics:XHunterAslFormat" mimetype="application/octet-stream" display_in_upload="true" />
|
||||
<datatype extension="idpdb" type="galaxy.datatypes.proteomics:IdpDB" display_in_upload="true" />
|
||||
<datatype extension="idpdb" type="galaxy.datatypes.binary:IdpDB" mimetype="application/octet-stream" display_in_upload="true" />
|
||||
<datatype extension="sf3" type="galaxy.datatypes.proteomics:Sf3" display_in_upload="true" />
|
||||
<datatype extension="cps" type="galaxy.datatypes.binary:Binary" subclass="True" display_in_upload="true" />
|
||||
<datatype extension="ct" type="galaxy.datatypes.tabular:ConnectivityTable" display_in_upload="True"/>
|
||||
@@ -446,8 +446,10 @@
|
||||
<sniffer type="galaxy.datatypes.tabular:Vcf"/>
|
||||
<sniffer type="galaxy.datatypes.binary:TwoBit"/>
|
||||
<sniffer type="galaxy.datatypes.binary:GeminiSQLite"/>
|
||||
<sniffer type="galaxy.datatypes.proteomics:MzSQlite"/>
|
||||
<sniffer type="galaxy.datatypes.binary:MzSQlite"/>
|
||||
<sniffer type="galaxy.datatypes.binary:IdpDB"/>
|
||||
<sniffer type="galaxy.datatypes.binary:SQlite"/>
|
||||
<sniffer type="galaxy.datatypes.binary:H5"/>
|
||||
<sniffer type="galaxy.datatypes.binary:Bam"/>
|
||||
<sniffer type="galaxy.datatypes.binary:CRAM"/>
|
||||
<sniffer type="galaxy.datatypes.binary:Sff"/>
|
||||
|
||||
@@ -540,12 +540,36 @@ Binary.register_sniffable_binary_format("bcf", "bcf", Bcf)
|
||||
|
||||
|
||||
class H5( Binary ):
|
||||
"""Class describing an HDF5 file"""
|
||||
"""
|
||||
Class describing an HDF5 file
|
||||
|
||||
>>> from galaxy.datatypes.sniff import get_test_fname
|
||||
>>> fname = get_test_fname( 'test.mz5' )
|
||||
>>> H5().sniff( fname )
|
||||
True
|
||||
>>> fname = get_test_fname( 'interval.interval' )
|
||||
>>> H5().sniff( fname )
|
||||
False
|
||||
"""
|
||||
file_ext = "h5"
|
||||
|
||||
def __init__( self, **kwd ):
|
||||
Binary.__init__( self, **kwd )
|
||||
self._magic = binascii.unhexlify("894844460d0a1a0a")
|
||||
|
||||
def sniff( self, filename ):
|
||||
# The first 8 bytes of any hdf5 file are 0x894844460d0a1a0a
|
||||
try:
|
||||
header = open( filename ).read(8)
|
||||
if header == self._magic:
|
||||
return True
|
||||
return False
|
||||
except:
|
||||
return False
|
||||
|
||||
def set_peek( self, dataset, is_multi_byte=False ):
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = "Binary h5 file"
|
||||
dataset.peek = "Binary HDF5 file"
|
||||
dataset.blurb = nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
@@ -555,9 +579,9 @@ class H5( Binary ):
|
||||
try:
|
||||
return dataset.peek
|
||||
except:
|
||||
return "Binary h5 sequence file (%s)" % ( nice_size( dataset.get_size() ) )
|
||||
return "Binary HDF5 file (%s)" % ( nice_size( dataset.get_size() ) )
|
||||
|
||||
Binary.register_unsniffable_binary_ext("h5")
|
||||
Binary.register_sniffable_binary_format("h5", "h5", H5)
|
||||
|
||||
|
||||
class Scf( Binary ):
|
||||
@@ -847,8 +871,84 @@ class GeminiSQLite( SQlite ):
|
||||
except:
|
||||
return "Gemini SQLite Database, version %s" % ( dataset.metadata.gemini_version or 'unknown' )
|
||||
|
||||
|
||||
class MzSQlite( SQlite ):
|
||||
"""Class describing a Proteomics Sqlite database """
|
||||
file_ext = "mz.sqlite"
|
||||
|
||||
def set_meta( self, dataset, overwrite=True, **kwd ):
|
||||
super( MzSQlite, self ).set_meta( dataset, overwrite=overwrite, **kwd )
|
||||
|
||||
def sniff( self, filename ):
|
||||
if super( MzSQlite, self ).sniff( filename ):
|
||||
mz_table_names = ["DBSequence", "Modification", "Peaks", "Peptide", "PeptideEvidence", "Score", "SearchDatabase", "Source", "SpectraData", "Spectrum", "SpectrumIdentification"]
|
||||
try:
|
||||
conn = sqlite.connect( filename )
|
||||
c = conn.cursor()
|
||||
tables_query = "SELECT name FROM sqlite_master WHERE type='table' ORDER BY name"
|
||||
result = c.execute( tables_query ).fetchall()
|
||||
result = map( lambda x: x[0], result )
|
||||
for table_name in mz_table_names:
|
||||
if table_name not in result:
|
||||
return False
|
||||
return True
|
||||
except Exception, e:
|
||||
log.warn( '%s, sniff Exception: %s', self, e )
|
||||
return False
|
||||
|
||||
|
||||
class IdpDB( SQlite ):
|
||||
"""
|
||||
Class describing an IDPicker 3 idpDB (sqlite) database
|
||||
|
||||
>>> from galaxy.datatypes.sniff import get_test_fname
|
||||
>>> fname = get_test_fname( 'test.idpDB' )
|
||||
>>> IdpDB().sniff( fname )
|
||||
True
|
||||
>>> fname = get_test_fname( 'interval.interval' )
|
||||
>>> IdpDB().sniff( fname )
|
||||
False
|
||||
"""
|
||||
file_ext = "idpdb"
|
||||
|
||||
def set_meta( self, dataset, overwrite=True, **kwd ):
|
||||
super( IdpDB, self ).set_meta( dataset, overwrite=overwrite, **kwd )
|
||||
|
||||
def sniff( self, filename ):
|
||||
if super( IdpDB, self ).sniff( filename ):
|
||||
mz_table_names = ["About", "Analysis", "AnalysisParameter", "PeptideSpectrumMatch", "Spectrum", "SpectrumSource"]
|
||||
try:
|
||||
conn = sqlite.connect( filename )
|
||||
c = conn.cursor()
|
||||
tables_query = "SELECT name FROM sqlite_master WHERE type='table' ORDER BY name"
|
||||
result = c.execute( tables_query ).fetchall()
|
||||
result = map( lambda x: x[0], result )
|
||||
for table_name in mz_table_names:
|
||||
if table_name not in result:
|
||||
return False
|
||||
return True
|
||||
except Exception, e:
|
||||
log.warn( '%s, sniff Exception: %s', self, e )
|
||||
return False
|
||||
|
||||
def set_peek( self, dataset, is_multi_byte=False ):
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = "IDPickerDB SQLite file"
|
||||
dataset.blurb = nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def display_peek( self, dataset ):
|
||||
try:
|
||||
return dataset.peek
|
||||
except:
|
||||
return "IDPickerDB SQLite file (%s)" % ( nice_size( dataset.get_size() ) )
|
||||
|
||||
Binary.register_sniffable_binary_format( "gemini.sqlite", "gemini.sqlite", GeminiSQLite )
|
||||
# FIXME: We need to register gemini.sqlite before sqlite, since register_sniffable_binary_format and is_sniffable_binary called in upload.py
|
||||
Binary.register_sniffable_binary_format( "idpdb", "idpdb", IdpDB )
|
||||
Binary.register_sniffable_binary_format( "mz.sqlite", "mz.sqlite", MzSQlite )
|
||||
# FIXME: We need to register specialized sqlite formats before sqlite, since register_sniffable_binary_format and is_sniffable_binary called in upload.py
|
||||
# ignores sniff order declared in datatypes_conf.xml
|
||||
Binary.register_sniffable_binary_format("sqlite", "sqlite", SQlite)
|
||||
|
||||
|
||||
@@ -6,11 +6,11 @@ import logging
|
||||
import re
|
||||
|
||||
from galaxy.datatypes import data
|
||||
from galaxy.datatypes.binary import Binary, SQlite
|
||||
from galaxy.datatypes.binary import Binary
|
||||
from galaxy.datatypes.data import Text
|
||||
from galaxy.datatypes.tabular import Tabular
|
||||
from galaxy.datatypes.xml import GenericXml
|
||||
from galaxy.util import nice_size, sqlite
|
||||
from galaxy.util import nice_size
|
||||
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -53,10 +53,6 @@ class Wiff(Binary):
|
||||
Binary.register_sniffable_binary_format("wiff", "wiff", Wiff )
|
||||
|
||||
|
||||
class IdpDB(Binary):
|
||||
file_ext = "idpDB"
|
||||
|
||||
|
||||
class PepXmlReport(Tabular):
|
||||
"""pepxml converted to tabular report"""
|
||||
file_ext = "tsv"
|
||||
@@ -406,28 +402,3 @@ class XHunterAslFormat(Binary):
|
||||
class Sf3(Binary):
|
||||
"""Class describing a Scaffold SF3 files"""
|
||||
file_ext = "sf3"
|
||||
|
||||
|
||||
class MzSQlite( SQlite ):
|
||||
"""Class describing a Proteomics Sqlite database """
|
||||
file_ext = "mz.sqlite"
|
||||
|
||||
def set_meta( self, dataset, overwrite=True, **kwd ):
|
||||
super( MzSQlite, self ).set_meta( dataset, overwrite=overwrite, **kwd )
|
||||
|
||||
def sniff( self, filename ):
|
||||
if super( MzSQlite, self ).sniff( filename ):
|
||||
mz_table_names = ["DBSequence", "Modification", "Peaks", "Peptide", "PeptideEvidence", "Score", "SearchDatabase", "Source", "SpectraData", "Spectrum", "SpectrumIdentification"]
|
||||
try:
|
||||
conn = sqlite.connect( filename )
|
||||
c = conn.cursor()
|
||||
tables_query = "SELECT name FROM sqlite_master WHERE type='table' ORDER BY name"
|
||||
result = c.execute( tables_query ).fetchall()
|
||||
result = map( lambda x: x[0], result )
|
||||
for table_name in mz_table_names:
|
||||
if table_name not in result:
|
||||
return False
|
||||
return True
|
||||
except Exception, e:
|
||||
log.warn( '%s, sniff Exception: %s', self, e )
|
||||
return False
|
||||
|
||||
@@ -651,18 +651,23 @@ class Registry( object ):
|
||||
'coverage' : coverage.LastzCoverage(),
|
||||
'customtrack' : interval.CustomTrack(),
|
||||
'csfasta' : sequence.csFasta(),
|
||||
'db3' : binary.SQlite(),
|
||||
'fasta' : sequence.Fasta(),
|
||||
'eland' : tabular.Eland(),
|
||||
'fastq' : sequence.Fastq(),
|
||||
'fastqsanger' : sequence.FastqSanger(),
|
||||
'gemini.sqlite' : binary.GeminiSQLite(),
|
||||
'gtf' : interval.Gtf(),
|
||||
'gff' : interval.Gff(),
|
||||
'gff3' : interval.Gff3(),
|
||||
'genetrack' : tracks.GeneTrack(),
|
||||
'h5' : binary.H5(),
|
||||
'idpdb' : binary.IdpDB(),
|
||||
'interval' : interval.Interval(),
|
||||
'laj' : images.Laj(),
|
||||
'lav' : sequence.Lav(),
|
||||
'maf' : sequence.Maf(),
|
||||
'mz.sqlite' : binary.MzSQlite(),
|
||||
'pileup' : tabular.Pileup(),
|
||||
'qualsolid' : qualityscore.QualityScoreSOLiD(),
|
||||
'qualsolexa' : qualityscore.QualityScoreSolexa(),
|
||||
@@ -684,18 +689,23 @@ class Registry( object ):
|
||||
'bed' : 'text/plain',
|
||||
'customtrack' : 'text/plain',
|
||||
'csfasta' : 'text/plain',
|
||||
'db3' : 'application/octet-stream',
|
||||
'eland' : 'application/octet-stream',
|
||||
'fasta' : 'text/plain',
|
||||
'fastq' : 'text/plain',
|
||||
'fastqsanger' : 'text/plain',
|
||||
'gemini.sqlite' : 'application/octet-stream',
|
||||
'gtf' : 'text/plain',
|
||||
'gff' : 'text/plain',
|
||||
'gff3' : 'text/plain',
|
||||
'h5' : 'application/octet-stream',
|
||||
'idpdb' : 'application/octet-stream',
|
||||
'interval' : 'text/plain',
|
||||
'laj' : 'text/plain',
|
||||
'lav' : 'text/plain',
|
||||
'maf' : 'text/plain',
|
||||
'memexml' : 'application/xml',
|
||||
'mz.sqlite' : 'application/octet-stream',
|
||||
'pileup' : 'text/plain',
|
||||
'qualsolid' : 'text/plain',
|
||||
'qualsolexa' : 'text/plain',
|
||||
@@ -720,6 +730,11 @@ class Registry( object ):
|
||||
self.sniff_order = [
|
||||
binary.Bam(),
|
||||
binary.Sff(),
|
||||
binary.H5(),
|
||||
binary.GeminiSQLite(),
|
||||
binary.MzSQlite(),
|
||||
binary.IdpDB(),
|
||||
binary.SQlite(),
|
||||
xml.GenericXml(),
|
||||
sequence.Maf(),
|
||||
sequence.Lav(),
|
||||
|
||||
@@ -310,6 +310,12 @@ def guess_ext( fname, sniff_order=None, is_multi_byte=False ):
|
||||
>>> fname = get_test_fname('3unsorted.bam')
|
||||
>>> guess_ext(fname)
|
||||
'bam'
|
||||
>>> fname = get_test_fname('test.idpDB')
|
||||
>>> guess_ext(fname)
|
||||
'idpdb'
|
||||
>>> fname = get_test_fname('test.mz5')
|
||||
>>> guess_ext(fname)
|
||||
'h5'
|
||||
"""
|
||||
if sniff_order is None:
|
||||
datatypes_registry = registry.Registry()
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Reference in New Issue
Block a user