Remove hard-coding of unsniffable binary types array and manually

checking each sniffable type with a seperate function in
upload.py. Information on both types is now stored dynamically as
static variables in the Binary class.
This commit is contained in:
John Chilton
2012-08-15 23:35:28 -05:00
parent 78e2d4f919
commit d281e835ef
4 changed files with 47 additions and 30 deletions
+36 -1
View File
@@ -18,10 +18,31 @@ import struct
log = logging.getLogger(__name__)
# Currently these supported binary data types must be manually set on upload
unsniffable_binary_formats = [ 'ab1', 'scf', 'h5' ]
class Binary( data.Data ):
"""Binary data"""
sniffable_binary_formats = []
unsniffable_binary_formats = []
@staticmethod
def register_sniffable_binary_format(data_type, ext, type_class):
Binary.sniffable_binary_formats.append({"type": data_type, "ext": ext, "class": type_class})
@staticmethod
def register_unsniffable_binary_ext(ext):
Binary.unsniffable_binary_formats.append(ext)
@staticmethod
def is_sniffable_binary(filename):
for format in Binary.sniffable_binary_formats:
if format["class"]().sniff(filename):
return (format["type"], format["ext"])
return None
@staticmethod
def is_ext_unsniffable(ext):
return ext in Binary.unsniffable_binary_formats
def set_peek( self, dataset, is_multi_byte=False ):
"""Set the peek and blurb text"""
if not dataset.dataset.purged:
@@ -62,6 +83,8 @@ class Ab1( Binary ):
except:
return "Binary ab1 sequence file (%s)" % ( data.nice_size( dataset.get_size() ) )
Binary.register_unsniffable_binary_ext("ab1")
class Bam( Binary ):
"""Class describing a BAM binary file"""
file_ext = "bam"
@@ -218,6 +241,8 @@ class Bam( Binary ):
def get_track_type( self ):
return "ReadTrack", {"data": "bai", "index": "summary_tree"}
Binary.register_sniffable_binary_format("bam", "bam", Bam)
class H5( Binary ):
"""Class describing an HDF5 file"""
file_ext = "h5"
@@ -235,6 +260,8 @@ class H5( Binary ):
except:
return "Binary h5 sequence file (%s)" % ( data.nice_size( dataset.get_size() ) )
Binary.register_unsniffable_binary_ext("h5")
class Scf( Binary ):
"""Class describing an scf binary sequence file"""
file_ext = "scf"
@@ -252,6 +279,8 @@ class Scf( Binary ):
except:
return "Binary scf sequence file (%s)" % ( data.nice_size( dataset.get_size() ) )
Binary.register_unsniffable_binary_ext("scf")
class Sff( Binary ):
""" Standard Flowgram Format (SFF) """
file_ext = "sff"
@@ -281,6 +310,8 @@ class Sff( Binary ):
except:
return "Binary sff file (%s)" % ( data.nice_size( dataset.get_size() ) )
Binary.register_sniffable_binary_format("sff", "sff", Sff)
class BigWig(Binary):
"""
Accessing binary BigWig files from UCSC.
@@ -314,6 +345,8 @@ class BigWig(Binary):
def get_track_type( self ):
return "LineTrack", {"data_standalone": "bigwig"}
Binary.register_sniffable_binary_format("bigwig", "bigwig", BigWig)
class BigBed(BigWig):
"""BigBed support from UCSC."""
def __init__( self, **kwd ):
@@ -323,6 +356,8 @@ class BigBed(BigWig):
def get_track_type( self ):
return "LineTrack", {"data_standalone": "bigbed"}
Binary.register_sniffable_binary_format("bigbed", "bigbed", BigBed)
class TwoBit (Binary):
"""Class describing a TwoBit format nucleotide file"""
+3
View File
@@ -4,6 +4,7 @@ Image classes
import data
import logging
from galaxy.datatypes.binary import Binary
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes import metadata
from galaxy.datatypes.sniff import *
@@ -154,6 +155,8 @@ class Pdf( Image ):
except IndexError:
return False
Binary.register_sniffable_binary_format("pdf", "pdf", Pdf)
def create_applet_tag_peek( class_name, archive, params ):
text = """
<!--[if !IE]>-->
+1 -2
View File
@@ -5,7 +5,6 @@ import logging, sys, os, csv, tempfile, shutil, re, zipfile, gzip
import registry
from galaxy import util
from galaxy.datatypes.checkers import *
from galaxy.datatypes.binary import unsniffable_binary_formats
log = logging.getLogger(__name__)
@@ -381,7 +380,7 @@ def handle_uploaded_dataset_file( filename, datatypes_registry, ext = 'auto', is
ext = guess_ext( filename, sniff_order = datatypes_registry.sniff_order, is_multi_byte=is_multi_byte )
if check_binary( filename ):
if ext not in unsniffable_binary_formats and not datatypes_registry.get_datatype_by_extension( ext ).sniff( filename ):
if not Binary.is_ext_unsniffable(ext) and not datatypes_registry.get_datatype_by_extension( ext ).sniff( filename ):
raise InappropriateDatasetContentError, 'The binary uploaded file contains inappropriate content.'
elif check_html( filename ):
raise InappropriateDatasetContentError, 'The uploaded file contains inappropriate HTML content.'
+7 -27
View File
@@ -58,16 +58,6 @@ def safe_dict(d):
return [safe_dict(x) for x in d]
else:
return d
def check_bam( file_path ):
return Bam().sniff( file_path )
def check_sff( file_path ):
return Sff().sniff( file_path )
def check_pdf( file_path ):
return Pdf().sniff( file_path )
def check_bigwig( file_path ):
return BigWig().sniff( file_path )
def check_bigbed( file_path ):
return BigBed().sniff( file_path )
def parse_outputs( args ):
rval = {}
for arg in args:
@@ -121,21 +111,11 @@ def add_file( dataset, registry, json_file, output_path ):
data_type = 'multi-byte char'
ext = sniff.guess_ext( dataset.path, is_multi_byte=True )
# Is dataset content supported sniffable binary?
elif check_bam( dataset.path ):
ext = 'bam'
data_type = 'bam'
elif check_sff( dataset.path ):
ext = 'sff'
data_type = 'sff'
elif check_pdf( dataset.path ):
ext = 'pdf'
data_type = 'pdf'
elif check_bigwig( dataset.path ):
ext = 'bigwig'
data_type = 'bigwig'
elif check_bigbed( dataset.path ):
ext = 'bigbed'
data_type = 'bigbed'
else:
type_info = Binary.is_sniffable_binary( dataset.path )
if type_info:
data_type = type_info[0]
ext = type_info[1]
if not data_type:
# See if we have a gzipped file, which, if it passes our restrictions, we'll uncompress
is_gzipped, is_valid = check_gzip( dataset.path )
@@ -267,10 +247,10 @@ def add_file( dataset, registry, json_file, output_path ):
parts = dataset.name.split( "." )
if len( parts ) > 1:
ext = parts[1].strip().lower()
if ext not in unsniffable_binary_formats:
if not Binary.is_ext_unsniffable(ext):
file_err( 'The uploaded binary file contains inappropriate content', dataset, json_file )
return
elif ext in unsniffable_binary_formats and dataset.file_type != ext:
elif Binary.is_ext_unsniffable(ext) and dataset.file_type != ext:
err_msg = "You must manually set the 'File Format' to '%s' when uploading %s files." % ( ext.capitalize(), ext )
file_err( err_msg, dataset, json_file )
return