diff --git a/config/datatypes_conf.xml.sample b/config/datatypes_conf.xml.sample index 1ab03308f5b..ed31bb126fd 100644 --- a/config/datatypes_conf.xml.sample +++ b/config/datatypes_conf.xml.sample @@ -107,7 +107,7 @@ - + @@ -168,7 +168,7 @@ - + @@ -177,7 +177,7 @@ - + @@ -446,8 +446,10 @@ - + + + diff --git a/lib/galaxy/datatypes/binary.py b/lib/galaxy/datatypes/binary.py index 3c8637ceb1c..fc26d8bd980 100644 --- a/lib/galaxy/datatypes/binary.py +++ b/lib/galaxy/datatypes/binary.py @@ -540,12 +540,36 @@ Binary.register_sniffable_binary_format("bcf", "bcf", Bcf) class H5( Binary ): - """Class describing an HDF5 file""" + """ + Class describing an HDF5 file + + >>> from galaxy.datatypes.sniff import get_test_fname + >>> fname = get_test_fname( 'test.mz5' ) + >>> H5().sniff( fname ) + True + >>> fname = get_test_fname( 'interval.interval' ) + >>> H5().sniff( fname ) + False + """ file_ext = "h5" + def __init__( self, **kwd ): + Binary.__init__( self, **kwd ) + self._magic = binascii.unhexlify("894844460d0a1a0a") + + def sniff( self, filename ): + # The first 8 bytes of any hdf5 file are 0x894844460d0a1a0a + try: + header = open( filename ).read(8) + if header == self._magic: + return True + return False + except: + return False + def set_peek( self, dataset, is_multi_byte=False ): if not dataset.dataset.purged: - dataset.peek = "Binary h5 file" + dataset.peek = "Binary HDF5 file" dataset.blurb = nice_size( dataset.get_size() ) else: dataset.peek = 'file does not exist' @@ -555,9 +579,9 @@ class H5( Binary ): try: return dataset.peek except: - return "Binary h5 sequence file (%s)" % ( nice_size( dataset.get_size() ) ) + return "Binary HDF5 file (%s)" % ( nice_size( dataset.get_size() ) ) -Binary.register_unsniffable_binary_ext("h5") +Binary.register_sniffable_binary_format("h5", "h5", H5) class Scf( Binary ): @@ -847,8 +871,84 @@ class GeminiSQLite( SQlite ): except: return "Gemini SQLite Database, version %s" % ( dataset.metadata.gemini_version or 'unknown' ) + +class MzSQlite( SQlite ): + """Class describing a Proteomics Sqlite database """ + file_ext = "mz.sqlite" + + def set_meta( self, dataset, overwrite=True, **kwd ): + super( MzSQlite, self ).set_meta( dataset, overwrite=overwrite, **kwd ) + + def sniff( self, filename ): + if super( MzSQlite, self ).sniff( filename ): + mz_table_names = ["DBSequence", "Modification", "Peaks", "Peptide", "PeptideEvidence", "Score", "SearchDatabase", "Source", "SpectraData", "Spectrum", "SpectrumIdentification"] + try: + conn = sqlite.connect( filename ) + c = conn.cursor() + tables_query = "SELECT name FROM sqlite_master WHERE type='table' ORDER BY name" + result = c.execute( tables_query ).fetchall() + result = map( lambda x: x[0], result ) + for table_name in mz_table_names: + if table_name not in result: + return False + return True + except Exception, e: + log.warn( '%s, sniff Exception: %s', self, e ) + return False + + +class IdpDB( SQlite ): + """ + Class describing an IDPicker 3 idpDB (sqlite) database + + >>> from galaxy.datatypes.sniff import get_test_fname + >>> fname = get_test_fname( 'test.idpDB' ) + >>> IdpDB().sniff( fname ) + True + >>> fname = get_test_fname( 'interval.interval' ) + >>> IdpDB().sniff( fname ) + False + """ + file_ext = "idpdb" + + def set_meta( self, dataset, overwrite=True, **kwd ): + super( IdpDB, self ).set_meta( dataset, overwrite=overwrite, **kwd ) + + def sniff( self, filename ): + if super( IdpDB, self ).sniff( filename ): + mz_table_names = ["About", "Analysis", "AnalysisParameter", "PeptideSpectrumMatch", "Spectrum", "SpectrumSource"] + try: + conn = sqlite.connect( filename ) + c = conn.cursor() + tables_query = "SELECT name FROM sqlite_master WHERE type='table' ORDER BY name" + result = c.execute( tables_query ).fetchall() + result = map( lambda x: x[0], result ) + for table_name in mz_table_names: + if table_name not in result: + return False + return True + except Exception, e: + log.warn( '%s, sniff Exception: %s', self, e ) + return False + + def set_peek( self, dataset, is_multi_byte=False ): + if not dataset.dataset.purged: + dataset.peek = "IDPickerDB SQLite file" + dataset.blurb = nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' + + def display_peek( self, dataset ): + try: + return dataset.peek + except: + return "IDPickerDB SQLite file (%s)" % ( nice_size( dataset.get_size() ) ) + Binary.register_sniffable_binary_format( "gemini.sqlite", "gemini.sqlite", GeminiSQLite ) -# FIXME: We need to register gemini.sqlite before sqlite, since register_sniffable_binary_format and is_sniffable_binary called in upload.py +Binary.register_sniffable_binary_format( "idpdb", "idpdb", IdpDB ) +Binary.register_sniffable_binary_format( "mz.sqlite", "mz.sqlite", MzSQlite ) +# FIXME: We need to register specialized sqlite formats before sqlite, since register_sniffable_binary_format and is_sniffable_binary called in upload.py # ignores sniff order declared in datatypes_conf.xml Binary.register_sniffable_binary_format("sqlite", "sqlite", SQlite) diff --git a/lib/galaxy/datatypes/proteomics.py b/lib/galaxy/datatypes/proteomics.py index c510e30ae46..da37108d348 100644 --- a/lib/galaxy/datatypes/proteomics.py +++ b/lib/galaxy/datatypes/proteomics.py @@ -6,11 +6,11 @@ import logging import re from galaxy.datatypes import data -from galaxy.datatypes.binary import Binary, SQlite +from galaxy.datatypes.binary import Binary from galaxy.datatypes.data import Text from galaxy.datatypes.tabular import Tabular from galaxy.datatypes.xml import GenericXml -from galaxy.util import nice_size, sqlite +from galaxy.util import nice_size log = logging.getLogger(__name__) @@ -53,10 +53,6 @@ class Wiff(Binary): Binary.register_sniffable_binary_format("wiff", "wiff", Wiff ) -class IdpDB(Binary): - file_ext = "idpDB" - - class PepXmlReport(Tabular): """pepxml converted to tabular report""" file_ext = "tsv" @@ -406,28 +402,3 @@ class XHunterAslFormat(Binary): class Sf3(Binary): """Class describing a Scaffold SF3 files""" file_ext = "sf3" - - -class MzSQlite( SQlite ): - """Class describing a Proteomics Sqlite database """ - file_ext = "mz.sqlite" - - def set_meta( self, dataset, overwrite=True, **kwd ): - super( MzSQlite, self ).set_meta( dataset, overwrite=overwrite, **kwd ) - - def sniff( self, filename ): - if super( MzSQlite, self ).sniff( filename ): - mz_table_names = ["DBSequence", "Modification", "Peaks", "Peptide", "PeptideEvidence", "Score", "SearchDatabase", "Source", "SpectraData", "Spectrum", "SpectrumIdentification"] - try: - conn = sqlite.connect( filename ) - c = conn.cursor() - tables_query = "SELECT name FROM sqlite_master WHERE type='table' ORDER BY name" - result = c.execute( tables_query ).fetchall() - result = map( lambda x: x[0], result ) - for table_name in mz_table_names: - if table_name not in result: - return False - return True - except Exception, e: - log.warn( '%s, sniff Exception: %s', self, e ) - return False diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index 843d1ec06e5..17c4b70c122 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -651,18 +651,23 @@ class Registry( object ): 'coverage' : coverage.LastzCoverage(), 'customtrack' : interval.CustomTrack(), 'csfasta' : sequence.csFasta(), + 'db3' : binary.SQlite(), 'fasta' : sequence.Fasta(), 'eland' : tabular.Eland(), 'fastq' : sequence.Fastq(), 'fastqsanger' : sequence.FastqSanger(), + 'gemini.sqlite' : binary.GeminiSQLite(), 'gtf' : interval.Gtf(), 'gff' : interval.Gff(), 'gff3' : interval.Gff3(), 'genetrack' : tracks.GeneTrack(), + 'h5' : binary.H5(), + 'idpdb' : binary.IdpDB(), 'interval' : interval.Interval(), 'laj' : images.Laj(), 'lav' : sequence.Lav(), 'maf' : sequence.Maf(), + 'mz.sqlite' : binary.MzSQlite(), 'pileup' : tabular.Pileup(), 'qualsolid' : qualityscore.QualityScoreSOLiD(), 'qualsolexa' : qualityscore.QualityScoreSolexa(), @@ -684,18 +689,23 @@ class Registry( object ): 'bed' : 'text/plain', 'customtrack' : 'text/plain', 'csfasta' : 'text/plain', + 'db3' : 'application/octet-stream', 'eland' : 'application/octet-stream', 'fasta' : 'text/plain', 'fastq' : 'text/plain', 'fastqsanger' : 'text/plain', + 'gemini.sqlite' : 'application/octet-stream', 'gtf' : 'text/plain', 'gff' : 'text/plain', 'gff3' : 'text/plain', + 'h5' : 'application/octet-stream', + 'idpdb' : 'application/octet-stream', 'interval' : 'text/plain', 'laj' : 'text/plain', 'lav' : 'text/plain', 'maf' : 'text/plain', 'memexml' : 'application/xml', + 'mz.sqlite' : 'application/octet-stream', 'pileup' : 'text/plain', 'qualsolid' : 'text/plain', 'qualsolexa' : 'text/plain', @@ -720,6 +730,11 @@ class Registry( object ): self.sniff_order = [ binary.Bam(), binary.Sff(), + binary.H5(), + binary.GeminiSQLite(), + binary.MzSQlite(), + binary.IdpDB(), + binary.SQlite(), xml.GenericXml(), sequence.Maf(), sequence.Lav(), diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index e0ae0c3bd5a..4509b095e59 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -310,6 +310,12 @@ def guess_ext( fname, sniff_order=None, is_multi_byte=False ): >>> fname = get_test_fname('3unsorted.bam') >>> guess_ext(fname) 'bam' + >>> fname = get_test_fname('test.idpDB') + >>> guess_ext(fname) + 'idpdb' + >>> fname = get_test_fname('test.mz5') + >>> guess_ext(fname) + 'h5' """ if sniff_order is None: datatypes_registry = registry.Registry() diff --git a/lib/galaxy/datatypes/test/test.idpDB b/lib/galaxy/datatypes/test/test.idpDB new file mode 100644 index 00000000000..140cb01d893 Binary files /dev/null and b/lib/galaxy/datatypes/test/test.idpDB differ diff --git a/lib/galaxy/datatypes/test/test.mz5 b/lib/galaxy/datatypes/test/test.mz5 new file mode 100644 index 00000000000..b43eba6ee31 Binary files /dev/null and b/lib/galaxy/datatypes/test/test.mz5 differ