diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index 74773c3f751..d51e1180c3e 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -33,12 +33,38 @@ class Data( object ): """ __metaclass__ = DataMeta - """Provide the set of display formats supported by this datatype """ - supported_display_apps = [] + """Stores the set of display applications, and viewing methods, supported by this datatype """ + supported_display_apps = {} + + def __init__(self, **kwd): + """Initialize the datatype""" + object.__init__(self, **kwd) + self.supported_display_apps = self.supported_display_apps.copy() + + def write_from_stream(self, dataset, stream): + """Writes data from a stream""" + fd = open(dataset.file_name, 'wb') + while 1: + chunk = stream.read(1048576) + if not chunk: + break + os.write(fd, chunk) + os.close(fd) + + def set_raw_data(self, dataset, data): + """Saves the data on the disc""" + fd = open(dataset.file_name, 'wb') + os.write(fd, data) + os.close(fd) + + def get_raw_data( self, dataset ): + """Returns the full data. To stream it open the file_name and read/write as needed""" + try: + return file(datset.file_name, 'rb').read(-1) + except OSError, e: + log.exception('%s reading a file that does not exist %s' % (self.__class__.__name__, dataset.file_name)) + return '' - def set_peek( self, dataset ): - dataset.peek = '' - dataset.blurb = 'data' def init_meta( self, dataset, copy_from=None ): # Metadata should be left mostly uninitialized. Dataset will # handle returning default values when metadata is not set. @@ -48,47 +74,91 @@ class Data( object ): # flag the object as modified for SQLAlchemy. if copy_from: dataset.metadata = copy_from.metadata + def set_meta( self, dataset ): + """Unimplemented method, allows guessing of metadata from contents of file""" + return True def missing_meta( self, dataset): + """Unimplemented method, Returns True if metadata is missing""" return False - def get_estimated_display_viewport( self, dataset ): - raise Exception( "'get_estimated_display_viewport' must be overridden in subclass." ) - def as_ucsc_display_file( self, dataset ): - raise Exception( "'as_ucsc_display_file' not supported for this datatype" ) - def as_gbrowse_display_file( self, dataset ): - raise Exception( "'as_gbrowse_display_file' not supported for this datatype" ) + def set_peek( self, dataset ): + """Set the peek and blurb text""" + dataset.peek = '' + dataset.blurb = 'data' def display_peek(self, dataset): + """Returns formated html of peek""" try: return escape(dataset.peek) except: return "peek unavailable" def display_name(self, dataset): + """Returns formated html of dataset name""" try: return escape(dataset.name) except: return "name unavailable" def display_info(self, dataset): + """Returns formated html of dataset info""" try: return escape(dataset.info) except: return "info unavailable" - def get_ucsc_sites(self, dataset): - return util.get_ucsc_by_build(dataset.dbkey) - def get_gbrowse_sites(self, dataset): - return util.get_gbrowse_sites_by_build(dataset.dbkey) def validate(self, dataset): """Unimplemented validate, return no exceptions""" return list() def repair_methods(self, dataset): """Unimplemented method, returns dict with method/option for repairing errors""" return None + def get_mime(self): + """Returns the mime type of the datatype""" + return 'application/octet-stream' + def add_display_app (self, app_id, label, file_function, links_function ): + """ + Adds a display app to the datatype. + app_id is a unique id + label is the primary display label, ie display at 'UCSC' + file_function is a string containing the name of the function that returns a properly formated display + links_function is a string containing the name of the function that returns a list of (link_name,link) + """ + self.supported_display_apps = self.supported_display_apps.copy() + self.supported_display_apps[app_id] = {'label':label,'file_function':file_function,'links_function':links_function} + def remove_display_app (self, app_id): + """Removes a display app from the datatype""" + self.supported_display_apps = self.supported_display_apps.copy() + try: + del self.supported_display_apps[app_id] + except: + log.exception('Tried to remove display app %s from datatype %s, but this display app is not declared.' % ( type, self.__class__.__name__ ) ) + def get_display_types(self): + """Returns display types available""" + return self.supported_display_apps.keys() + def get_display_label(self, type): + """Returns primary label for display app""" + try: + return self.supported_display_apps[type]['label'] + except: + return 'unknown' + def as_display_type(self, dataset, type, **kwd): + """Returns modified file contents for a particular display type """ + try: + if type in self.get_display_types(): + return getattr (self, self.supported_display_apps[type]['file_function']) (dataset, **kwd) + except: + log.exception('Function %s is referred to in datatype %s for displaying as type %s, but is not accessible' % (self.supported_display_apps[type]['file_function'], self.__class__.__name__, type) ) + return "This display type (%s) is not implemented for this datatype (%s)." % ( type, dataset.ext) + + def get_display_links(self, dataset, type, app, base_url, **kwd): + """Returns a list of tuples of (name, link) for a particular display type """ + try: + if type in self.get_display_types(): + return getattr (self, self.supported_display_apps[type]['links_function']) (dataset, type, app, base_url, **kwd) + except: + log.exception('Function %s is referred to in datatype %s for generating links for type %s, but is not accessible' % (self.supported_display_apps[type]['links_function'], self.__class__.__name__, type) ) + return [] class Text( Data ): - """Provide the set of display formats supported by this datatype """ - supported_display_apps = [] - - def write_from_stream(self, stream): - "Writes data from a stream" + def write_from_stream(self, dataset, stream): + """Writes data from a stream""" # write it twice for now fd, temp_name = tempfile.mkstemp() while 1: @@ -99,39 +169,30 @@ class Text( Data ): os.close(fd) # rewrite the file with unix newlines - fp = open(self.file_name, 'wt') + fp = open(dataset.file_name, 'wt') for line in file(temp_name, "U"): line = line.strip() + '\n' fp.write(line) fp.close() - def set_raw_data(self, data): + def set_raw_data(self, dataset, data): """Saves the data on the disc""" fd, temp_name = tempfile.mkstemp() os.write(fd, data) os.close(fd) # rewrite the file with unix newlines - fp = open(self.file_name, 'wt') + fp = open(dataset.file_name, 'wt') for line in file(temp_name, "U"): line = line.strip() + '\n' fp.write(line) fp.close() os.remove( temp_name ) - - def delete(self): - """Remove the file that corresponds to this data""" - obj.DBObj.delete(self) - try: - os.remove(self.file_name) - except OSError, e: - log.critical('%s delete error %s' % (self.__class__.__name__, e)) - -# Removed for now ... this should be handled specifically by the registry -## def get_mime(self): -## """Returns the mime type of the data""" -## return galaxy.datatypes.registry.Registry().get_mimetype_by_extension( self.extension.lower() ) + + def get_mime(self): + """Returns the mime type of the datatype""" + return 'text/plain' def set_peek(self, dataset): dataset.peek = get_file_peek( dataset.file_name ) @@ -140,10 +201,8 @@ class Text( Data ): class Binary( Data ): """Binary data""" - """Provide the set of display formats supported by this datatype """ - supported_display_apps = [] - def set_peek( self, dataset ): + """Set the peek and blurb text""" dataset.peek = 'binary data' dataset.blurb = 'data' diff --git a/lib/galaxy/datatypes/images.py b/lib/galaxy/datatypes/images.py index 5cab3a93d1a..7ba9182948a 100644 --- a/lib/galaxy/datatypes/images.py +++ b/lib/galaxy/datatypes/images.py @@ -14,6 +14,7 @@ class Image( data.Data ): dataset.peek = 'Image in %s format (%s)' % ( dataset.extension, data.nice_size( dataset.get_size() ) ) dataset.blurb = 'image' + class Gmaj( data.Data ): """Class describing a GMAJ Applet""" def set_peek( self, dataset ): @@ -25,7 +26,11 @@ class Gmaj( data.Data ): return dataset.peek except: return "peek unavailable" - + def get_mime(self): + """Returns the mime type of the datatype""" + return 'application/zip' + + class Laj( data.Text ): """Class describing a LAJ Applet""" def set_peek( self, dataset ): @@ -45,8 +50,6 @@ class Html( data.Text ): dataset.peek = "HTML file (%s)" % ( data.nice_size( dataset.get_size() ) ) dataset.blurb = data.nice_size( dataset.get_size() ) - def display_peek(self, dataset): - try: - return dataset.peek - except: - return "peek unavailable" \ No newline at end of file + def get_mime(self): + """Returns the mime type of the datatype""" + return 'text/html' \ No newline at end of file diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index ad77930bfff..478e75b5487 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -9,6 +9,7 @@ import logging, os, sys, time, sets, tempfile, shutil import data from galaxy import util from cgi import escape +import urllib from bx.intervals.io import * from galaxy.datatypes.metadata import MetadataElement from galaxy.datatypes.tabular import Tabular @@ -35,15 +36,18 @@ for key, value in alias_spec.items(): class Interval( Tabular ): """Tab delimited data containing interval information""" - """Provide the set of display formats supported by this datatype """ - supported_display_apps = ['ucsc'] - """Add metadata elements""" MetadataElement( name="chromCol" ) MetadataElement( name="startCol" ) MetadataElement( name="endCol" ) MetadataElement( name="strandCol" ) + + def __init__(self, **kwd): + """Initialize interval datatype, by adding UCSC display apps""" + Tabular.__init__(self, **kwd) + self.add_display_app ( 'ucsc', 'display at UCSC', 'as_ucsc_display_file', 'ucsc_links' ) + def missing_meta( self, dataset ): """Checks for empty meta values""" for key, value in dataset.metadata.items(): @@ -52,17 +56,18 @@ class Interval( Tabular ): return True return False - def init_meta( self, dataset, copy_from=None ): Tabular.init_meta( self, dataset, copy_from=copy_from ) for key in alias_spec: setattr( dataset.metadata, key, '' ) - setattr( dataset.metadata, 'strandCol', '0' ) + setattr( dataset.metadata, 'strandCol', '0' ) def set_peek( self, dataset ): + """Set the peek and blurb text""" dataset.peek = data.get_file_peek( dataset.file_name ) ## dataset.peek = self.make_html_table( dataset.peek ) dataset.blurb = util.commaify( str( data.get_line_count( dataset.file_name ) ) ) + " regions" + #i don't think set_meta should not be called here, it should be called separately self.set_meta( dataset ) def set_meta( self, dataset, first_line_is_header=False ): @@ -85,7 +90,7 @@ class Interval( Tabular ): for lower in values[start:]: del valid[lower] # removes lower priority keys dataset.mark_metadata_changed() - + def get_estimated_display_viewport( self, dataset ): """Return a chrom, start, stop tuple for viewing a file.""" if dataset.has_data() and dataset.state == dataset.states.OK: @@ -107,14 +112,14 @@ class Interval( Tabular ): start = min( start, int( p[s] ) ) stop = max( stop, int( p[e] ) ) except Exception, exc: - log.error( 'Viewport generation error -> %s ' % str(exc) ) + #log.error( 'Viewport generation error -> %s ' % str(exc) ) (chr, start, stop) = 'chr1', 1, 1000 return (chr, str( start ), str( stop )) else: return ('', '', '') - def as_ucsc_display_file( self, dataset ): - """Returns a file that contains only the bed data""" + def as_ucsc_display_file( self, dataset, **kwd ): + """Returns file contents with only the bed data""" fd, temp_name = tempfile.mkstemp() c, s, e, t = dataset.metadata.chromCol, dataset.metadata.startCol, dataset.metadata.endCol, dataset.metadata.strandCol c, s, e, t = int(c)-1, int(s)-1, int(e)-1, int(t)-1 @@ -129,7 +134,22 @@ class Interval( Tabular ): tmp = [ elems[c], elems[s], elems[e] ] os.write(fd, '%s\n' % '\t'.join(tmp) ) os.close(fd) - return temp_name + return open(temp_name) + + def ucsc_links( self, dataset, type, app, base_url ): + ret_val = [] + if dataset.has_data: + viewport_tuple = self.get_estimated_display_viewport(dataset) + if viewport_tuple: + chrom = viewport_tuple[0] + start = viewport_tuple[1] + stop = viewport_tuple[2] + for site_name, site_url in util.get_ucsc_by_build(dataset.dbkey): + if site_name in app.config.ucsc_display_sites: + display_url = urllib.quote_plus( "%s/display_as?id=%i&display_app=%s" % (base_url, dataset.id, type) ) + link = "%sdb=%s&position=%s:%s-%s&hgt.customText=%s" % (site_url, dataset.dbkey, chrom, start, stop, display_url ) + ret_val.append( (site_name, link) ) + return ret_val def validate( self, dataset ): """Validate an interval file using the bx GenomicIntervalReader""" @@ -160,10 +180,6 @@ class Interval( Tabular ): class Bed( Interval ): """Tab delimited data in BED format""" - """Provide the set of display formats supported by this datatype """ - supported_display_apps = ['ucsc'] - - """Add metadata elements""" MetadataElement( name="chromCol", default=1 ) MetadataElement( name="startCol", default=2 ) @@ -176,7 +192,7 @@ class Bed( Interval ): def init_meta( self, dataset, copy_from=None ): Interval.init_meta( self, dataset, copy_from=copy_from ) - + def set_meta( self, dataset ): """ Overrides the default setting for dataset.metadata.strandCol for BED @@ -200,8 +216,8 @@ class Bed( Interval ): dataset.metadata.strandCol = 0 dataset.mark_metadata_changed() - def as_ucsc_display_file( self, dataset ): - """Returns a file that contains only the bed data. If bed 6+, treat as interval.""" + def as_ucsc_display_file( self, dataset, **kwd ): + """Returns file contents with only the bed data. If bed 6+, treat as interval.""" for line in open(dataset.file_name): line = line.strip() if line == "" or line.startswith("#"): @@ -232,29 +248,24 @@ class Bed( Interval ): #only check first line for proper form break - try: return dataset.file_name + try: return open(dataset.file_name) except: return "This item contains no content" - def get_estimated_display_viewport( self, dataset ): - #TODO: fix me... - return Interval.get_estimated_display_viewport( self, dataset ) - class Gff( Tabular ): """Tab delimited data in Gff format""" - """Provide the set of display formats supported by this datatype """ - supported_display_apps = ['gbrowse'] - - def __init__(self, id=None): - data.Text.__init__(self, id=id) + def __init__(self, **kwd): + """Initialize datatype, by adding GBrowse display app""" + Tabular.__init__(self, **kwd) + self.add_display_app ('gbrowse', 'display in GBrowse', 'as_gbrowse_display_file', 'gbrowse_links' ) def make_html_table(self, data): return Tabular.make_html_table(self, data, skipchar='#') - def as_gbrowse_display_file( self, dataset ): - '''Returns a file that can be displayed in GBrowse apps.''' + def as_gbrowse_display_file( self, dataset, **kwd ): + """Returns file contents that can be displayed in GBrowse apps.""" #TODO: fix me... - return dataset.file_name + return open(dataset.file_name) def get_estimated_display_viewport( self, dataset ): """ @@ -289,42 +300,45 @@ class Gff( Tabular ): start = min( start, int( p[start_col] ) ) stop = max( stop, int( p[stop_col] ) ) except Exception, exc: - log.error( 'Viewport generation error -> %s ' % str(exc) ) + #log.error( 'Viewport generation error -> %s ' % str(exc) ) seqid, start, stop = ('', '', '') return (seqid, str( start ), str( stop )) else: return ('', '', '') + def gbrowse_links( self, dataset, type, app, base_url ): + ret_val = [] + if dataset.has_data: + viewport_tuple = self.get_estimated_display_viewport(dataset) + if viewport_tuple: + chrom = viewport_tuple[0] + start = viewport_tuple[1] + stop = viewport_tuple[2] + for site_name, site_url in util.get_gbrowse_sites_by_build(dataset.dbkey): + if site_name in app.config.gbrowse_display_sites: + display_url = urllib.quote_plus( "%s/display_as?id=%i&display_app=%s" % (base_url, dataset.id, type) ) + link = "%sname=%s&ref=%s:%s..%s&eurl=%s" % (site_url, dataset.dbkey, chrom, start, stop, display_url ) + ret_val.append( (site_name, link) ) + return ret_val + class Wiggle( Tabular ): """Tab delimited data in wiggle format""" - - """Provide the set of display formats supported by this datatype """ - supported_display_apps = [] - - def __init__(self, id=None): - data.Text.__init__(self, id=id) def make_html_table(self, data): return Tabular.make_html_table(self, data, skipchar='#') - - def get_estimated_display_viewport( self, dataset ): - #TODO: fix me... - return ('', '', '') #Extend Tabular type, since interval tools will fail on track def line (we should fix this) #This is a skeleton class for now, allows viewing at ucsc and formatted peeking. class CustomTrack ( Tabular ): """UCSC CustomTrack""" - - """Provide the set of display formats supported by this datatype """ - supported_display_apps = ['ucsc'] - - def __init__(self, id=None): - data.Text.__init__(self, id=id) + + def __init__(self, **kwd): + """Initialize interval datatype, by adding UCSC display app""" + Tabular.__init__(self, **kwd) + self.add_display_app ( 'ucsc', 'display at UCSC', 'as_ucsc_display_file', 'ucsc_links' ) def make_html_table(self, dataset): return Tabular.make_html_table(self, dataset, skipchar='track') - def get_estimated_display_viewport( self, dataset ): try: for line in open(dataset.file_name): @@ -344,28 +358,62 @@ class CustomTrack ( Tabular ): return ('', '', '') def as_ucsc_display_file( self, dataset ): - return dataset.file_name + return open(dataset.file_name) + def ucsc_links( self, dataset, type, app, base_url ): + ret_val = [] + if dataset.has_data: + viewport_tuple = self.get_estimated_display_viewport(dataset) + if viewport_tuple: + chrom = viewport_tuple[0] + start = viewport_tuple[1] + stop = viewport_tuple[2] + for site_name, site_url in util.get_ucsc_by_build(dataset.dbkey): + if site_name in app.config.ucsc_display_sites: + display_url = urllib.quote_plus( "%s/display_as?id=%i&display_app=%s" % (base_url, dataset.id, type) ) + link = "%sdb=%s&position=%s:%s-%s&hgt.customText=%s" % (site_url, dataset.dbkey, chrom, start, stop, display_url ) + ret_val.append( (site_name, link) ) + return ret_val + + + #Extend Tabular type, since interval tools will fail on track def line (we should fix this) -#This is a skeleton class for now, allows viewing at ucsc and formatted peeking. +#This is a skeleton class for now, allows viewing at GBrowse and formatted peeking. class GBrowseTrack ( Tabular ): - """Provide the set of display formats supported by this datatype """ - supported_display_apps = ['gbrowse'] + def __init__(self, **kwd): + """Initialize datatype, by adding GBrowse display app""" + Tabular.__init__(self, **kwd) + self.add_display_app ('gbrowse', 'display in GBrowse', 'as_gbrowse_display_file', 'gbrowse_links' ) - def __init__(self, id=None): - data.Text.__init__(self, id=id) def make_html_table(self, dataset): return Tabular.make_html_table(self, dataset, skipchar='track') - def display_formats_supported( self, dataset ): - return set(['gbrowse track']) - def get_estimated_display_viewport( self, dataset ): #TODO: fix me... return ('', '', '') + + def gbrowse_links( self, dataset, type, app, base_url ): + ret_val = [] + if dataset.has_data: + viewport_tuple = self.get_estimated_display_viewport(dataset) + if viewport_tuple: + chrom = viewport_tuple[0] + start = viewport_tuple[1] + stop = viewport_tuple[2] + for site_name, site_url in util.get_gbrowse_sites_by_build(dataset.dbkey): + if site_name in app.config.gbrowse_display_sites: + display_url = urllib.quote_plus( "%s/display_as?id=%i&display_app=%s" % (base_url, dataset.id, type) ) + link = "%sname=%s&ref=%s:%s..%s&eurl=%s" % (site_url, dataset.dbkey, chrom, start, stop, display_url ) + ret_val.append( (site_name, link) ) + return ret_val + + def as_gbrowse_display_file( self, dataset, **kwd ): + """Returns file contents that can be displayed in GBrowse apps.""" + #TODO: fix me... + return open(dataset.file_name) if __name__ == '__main__': import doctest, sys - doctest.testmod(sys.modules[__name__]) + doctest.testmod(sys.modules[__name__]) diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index 8ddd7a0b44b..ddbd17b9239 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -11,7 +11,7 @@ class Registry( object ): self.mimetypes_by_extension = {} for ext, kind in datatypes: try: - mime_type = 'text/plain' #default type of plain text + mime_type = None fields = kind.split(",") if len(fields)>1: kind = fields[0].strip() @@ -23,6 +23,9 @@ class Registry( object ): module = __import__(fields.pop(0)) for mod in fields: module = getattr(module,mod) self.datatypes_by_extension[ext] = getattr(module, datatype_class)() + if mime_type is None: + # Use default mime type as per datatype spec + mime_type = self.datatypes_by_extension[ext].get_mime() self.mimetypes_by_extension[ext] = mime_type except: self.log.warning('error loading datatype: %s' % ext) diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index c0fed65b60a..3c750ac70e7 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -9,16 +9,12 @@ log = logging.getLogger(__name__) class Sequence( data.Text ): """Class describing a sequence""" + pass - """Provide the set of display formats supported by this datatype """ - supported_display_apps = [] class Fasta( Sequence ): """Class representing a FASTA sequence""" - """Provide the set of display formats supported by this datatype """ - supported_display_apps = [] - def set_peek( self, dataset ): Sequence.set_peek( self, dataset ) count = size = 0 @@ -33,24 +29,14 @@ class Fasta( Sequence ): else: dataset.blurb = '%d sequences' % count - def get_estimated_display_viewport( self, dataset ): - #TODO: fix me... - return ('', '', '') - class Maf( Sequence ): """Class describing a Maf alignment""" - - """Provide the set of display formats supported by this datatype """ - supported_display_apps = [] + pass class Axt( Sequence ): """Class describing an axt alignment""" - - """Provide the set of display formats supported by this datatype """ - supported_display_apps = [] - + pass + class Lav( Sequence ): """Class describing a LAV alignment""" - - """Provide the set of display formats supported by this datatype """ - supported_display_apps = [] \ No newline at end of file + pass diff --git a/lib/galaxy/datatypes/tabular.py b/lib/galaxy/datatypes/tabular.py index 927912552de..9af1ba1c125 100644 --- a/lib/galaxy/datatypes/tabular.py +++ b/lib/galaxy/datatypes/tabular.py @@ -11,16 +11,12 @@ from galaxy import util from cgi import escape from galaxy.datatypes.metadata import MetadataElement from galaxy.datatypes.metadata import MetadataAttributes - log = logging.getLogger(__name__) class Tabular( data.Text ): """Tab delimited data""" - """Provide the set of display formats supported by this datatype """ - supported_display_apps = [] - MetadataElement( name="columns", default=0, desc="Number of columns", @@ -43,7 +39,6 @@ class Tabular( data.Text ): setattr( dataset.metadata, "columns", maxcols ) except: pass - def missing_meta( self, dataset ): """Checks for empty meta values""" for key, value in dataset.metadata.items(): @@ -52,6 +47,7 @@ class Tabular( data.Text ): return False def make_html_table(self, data, skipchar=None): + """Create HTML table, used for displaying peek""" out = ['