diff --git a/lib/galaxy/app.py b/lib/galaxy/app.py index 14ee687626b..49d718640f1 100644 --- a/lib/galaxy/app.py +++ b/lib/galaxy/app.py @@ -14,7 +14,7 @@ class UniverseApplication( object ): self.config.check() config.configure_logging( self.config ) #Set up datatypes registry - self.datatypes_registry = galaxy.datatypes.registry.Registry(datatypes = self.config.datatypes) + self.datatypes_registry = galaxy.datatypes.registry.Registry(datatypes=self.config.datatypes, sniff_order=self.config.sniff_order) galaxy.model.set_datatypes_registry(self.datatypes_registry) # Connect up the object model if self.config.database_connection: diff --git a/lib/galaxy/config.py b/lib/galaxy/config.py index 184ec202913..26c4303513d 100644 --- a/lib/galaxy/config.py +++ b/lib/galaxy/config.py @@ -60,6 +60,11 @@ class Configuration( object ): self.datatypes = global_conf_parser.items("galaxy:datatypes") except ConfigParser.NoSectionError: self.datatypes = [] + #Store sniff order config + try: + self.sniff_order = global_conf_parser.items("galaxy:sniff_order") + except ConfigParser.NoSectionError: + self.sniff_order = [] self.datatype_converters_config = kwargs.get( 'datatype_converters_config_file', "datatype_converters_conf.xml" ) self.datatype_converters_path = kwargs.get( 'datatype_converters_path', os.path.join(self.root,"lib/galaxy/datatypes/converters") ) def get( self, key, default ): diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index 745b8f423c3..19c9d333a3c 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -2,10 +2,14 @@ import logging, os, sys, time, sets, tempfile from galaxy import util from cgi import escape from galaxy.datatypes.metadata import * -log = logging.getLogger(__name__) from galaxy.datatypes.metadata import MetadataElement from galaxy.datatypes import metadata +log = logging.getLogger(__name__) + +# Valid strand column values +valid_strand = ['+', '-', '.'] + # Constants for data states DATA_NEW, DATA_OK, DATA_FAKE = 'new', 'ok', 'fake' diff --git a/lib/galaxy/datatypes/images.py b/lib/galaxy/datatypes/images.py index bdd8d3eebb4..edf8f696ea4 100644 --- a/lib/galaxy/datatypes/images.py +++ b/lib/galaxy/datatypes/images.py @@ -4,6 +4,7 @@ Image classes import data import logging +from galaxy.datatypes.sniff import * from urllib import urlencode log = logging.getLogger(__name__) @@ -17,6 +18,8 @@ class Image( data.Data ): class Gmaj( data.Data ): """Class describing a GMAJ Applet""" + file_ext = "gmaj.zip" + def set_peek( self, dataset ): dataset.peek = "

Your browser is not responding to the <applet> tag.

" dataset.blurb = 'GMAJ Multiple Alignment Viewer' @@ -29,27 +32,57 @@ class Gmaj( data.Data ): def get_mime(self): """Returns the mime type of the datatype""" return 'application/zip' + def sniff( self, filename ): + #TODO: fix me + return '' +class Html( data.Text ): + """Class describing an html file""" + file_ext = "html" + + def set_peek( self, dataset ): + dataset.peek = "HTML file (%s)" % ( data.nice_size( dataset.get_size() ) ) + dataset.blurb = data.nice_size( dataset.get_size() ) + + def get_mime(self): + """Returns the mime type of the datatype""" + return 'text/html' + + def sniff( self, filename ): + """ + Determines wether the file is in html format + + >>> fname = get_test_fname( 'complete.bed' ) + >>> Html().sniff( fname ) + '' + >>> fname = get_test_fname( 'file.html' ) + >>> Html().sniff( fname ) + 'html' + """ + headers = get_headers( filename, None ) + + try: + for i, hdr in enumerate(headers): + if hdr and hdr[0].lower().find( '' ) >=0: + return self.file_ext + return '' + except: + return '' class Laj( data.Text ): """Class describing a LAJ Applet""" + file_ext = "laj" + def set_peek( self, dataset ): export_url = "/history_add_to?"+urlencode({'history_id':dataset.history_id,'ext':'lav','name':'LAJ Output','info':'Added by LAJ','dbkey':dataset.dbkey}) dataset.peek = "

" dataset.blurb = 'LAJ Multiple Alignment Viewer' - def display_peek(self, dataset): try: return dataset.peek except: return "peek unavailable" - -class Html( data.Text ): - """Class describing an html file""" - def set_peek( self, dataset ): - dataset.peek = "HTML file (%s)" % ( data.nice_size( dataset.get_size() ) ) - dataset.blurb = data.nice_size( dataset.get_size() ) - - def get_mime(self): - """Returns the mime type of the datatype""" - return 'text/html' \ No newline at end of file + def sniff( self, filename ): + #TODO: fix me... + return '' + diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index 12554fccfd6..4b73bb268b1 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -8,6 +8,7 @@ pkg_resources.require( "bx-python" ) import logging, os, sys, time, sets, tempfile, shutil import data from galaxy import util +from galaxy.datatypes.sniff import * from cgi import escape import urllib from bx.intervals.io import * @@ -36,6 +37,7 @@ for key, value in alias_spec.items(): class Interval( Tabular ): """Tab delimited data containing interval information""" + file_ext = "interval" """Add metadata elements""" MetadataElement( name="chromCol", desc="Chrom column", param=metadata.ColumnParameter ) @@ -44,7 +46,6 @@ class Interval( Tabular ): MetadataElement( name="strandCol", desc="Strand column (click box & select)", param=metadata.ColumnParameter, optional=True, no_value=0 ) MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True, visible=False ) - def __init__(self, **kwd): """Initialize interval datatype, by adding UCSC display apps""" Tabular.__init__(self, **kwd) @@ -179,8 +180,41 @@ class Interval( Tabular ): """Return options for removing errors along with a description""" return [("lines","Remove erroneous lines")] + def sniff( self, filename ): + """ + Checks for 'intervalness' + + This format is mostly used by galaxy itself. Valid interval files should include + a valid header comment, but this seems to be loosely regulated. + + >>> fname = get_test_fname( 'test_space.bed' ) + >>> Interval().sniff( fname ) + '' + >>> fname = get_test_fname( 'interval.interval' ) + >>> Interval().sniff( fname ) + 'interval' + """ + headers = get_headers( filename, '\t' ) + try: + """ + If we got here, we already know the file is_column_based and is not bed, + so we'll just look for some valid data. + """ + for hdr in headers: + if not (hdr[0] == '' or hdr[0].startswith( '#' )): + if len(hdr) < 3: + return '' + try: + map( int, [hdr[1], hdr[2]] ) + except: + return '' + return self.file_ext + except: + return '' + class Bed( Interval ): """Tab delimited data in BED format""" + file_ext = "bed" """Add metadata elements""" MetadataElement( name="chromCol", default=1, desc="Chrom column", param=metadata.ColumnParameter ) @@ -264,8 +298,93 @@ class Bed( Interval ): try: return open(dataset.file_name) except: return "This item contains no content" + def sniff( self, filename ): + """ + Checks for 'bedness' + + BED lines have three required fields and nine additional optional fields. + The number of fields per line must be consistent throughout any single set of data in + an annotation track. + + For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format1 + + >>> fname = get_test_fname( 'test_tab.bed' ) + >>> Bed().sniff( fname ) + 'bed' + >>> fname = get_test_fname( 'interval.bed' ) + >>> Bed().sniff( fname ) + '' + >>> fname = get_test_fname( 'complete.bed' ) + >>> Bed().sniff( fname ) + 'bed' + """ + col1_startswith = ['chr', 'chl', 'groupun', 'reftig_', 'scaffold', 'super_', 'vcho'] + headers = get_headers( filename, '\t' ) + try: + if not headers: + return '' + for hdr in headers: + valid_col1 = False + if len(hdr) < 3 or len(hdr) > 12: + return '' + for str in col1_startswith: + if hdr[0].lower().startswith(str): + valid_col1 = True + break + if valid_col1: + try: + map( int, [hdr[1], hdr[2]] ) + except: + return '' + if len(hdr) > 3: + """ + Since all 9 of these fields are optional, it is difficult to test + for specific column values... + """ + optionals = hdr[3:] + """ + ...we can, however, test complete BED definitions fairly easily. + """ + if len(optionals) == 9: + try: + map ( int, [optionals[1], optionals[3], optionals[4], optionals[5], optionals[6]] ) + except: + return '' + score = int(optionals[1]) + if score < 0 or score > 1000: + return '' + if optionals[2] not in ['+', '-']: + return '' + if int(optionals[5]) != 0: + return '' + block_count = int(optionals[6]) + """ + Sometimes the blosck_sizes and block_starts lists end in extra commas + """ + block_sizes = optionals[7].rstrip(',').split(',') + block_starts = optionals[8].rstrip(',').split(',') + if len(block_sizes) != block_count or len(block_starts) != block_count: + return '' + elif len(optionals) > 4 and len(optionals) < 9: + """ + Here it gets a bit trickier, but in this case, we can be somewhat confident + that optionals will include a strand column + """ + is_valid_strand = False + for ele in optionals: + if ele in data.valid_strand: + is_valid_strand = True + if not is_valid_strand: + return '' + else: + return '' + return self.file_ext + except: + return '' + class Gff( Tabular ): """Tab delimited data in Gff format""" + file_ext = "gff" """Add metadata elements""" MetadataElement( name="columns", default=9, desc="Number of columns", readonly=True ) @@ -330,8 +449,51 @@ class Gff( Tabular ): ret_val.append( (site_name, link) ) return ret_val + def sniff( self, filename ): + """ + Determines whether the file is in gff format + + GFF lines have nine required fields that must be tab-separated. + + For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format3 + + >>> fname = get_test_fname( 'gff_version_3.gff' ) + >>> Gff().sniff( fname ) + '' + >>> fname = get_test_fname( 'test.gff' ) + >>> Gff().sniff( fname ) + 'gff' + """ + headers = get_headers( filename, '\t' ) + try: + if len(headers) < 2: + return '' + for hdr in headers: + if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '2' ) < 0: + return '' + if hdr and hdr[0] and not hdr[0].startswith( '#' ): + if len(hdr) != 9: + return '' + try: + map( int, [hdr[3], hdr[4]] ) + except: + return '' + if hdr[5] != '.': + try: + score = int(hdr[5]) + except: + return '' + if (score < 0 or score > 1000): + return '' + if hdr[6] not in data.valid_strand: + return '' + return self.file_ext + except: + return '' + class Gff3( Gff ): """Tab delimited data in Gff3 format""" + file_ext = "gff3" """Add metadata elements""" MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], desc="Column types", readonly=True ) @@ -340,18 +502,116 @@ class Gff3( Gff ): """Initialize datatype, by adding GBrowse display app""" Gff.__init__(self, **kwd) + def sniff( self, filename ): + """ + Determines whether the file is in gff version 3 format + + GFF 3 format: + + 1) adds a mechanism for representing more than one level + of hierarchical grouping of features and subfeatures. + 2) separates the ideas of group membership and feature name/id + 3) constrains the feature type field to be taken from a controlled + vocabulary. + 4) allows a single feature, such as an exon, to belong to more than + one group at a time. + 5) provides an explicit convention for pairwise alignments + 6) provides an explicit convention for features that occupy disjunct regions + + The format consists of 9 columns, separated by tabs (NOT spaces). + + Undefined fields are replaced with the "." character, as described in the original GFF spec. + + For complete details see http://song.sourceforge.net/gff3.shtml + + >>> fname = get_test_fname( 'test.gff' ) + >>> Gff3().sniff( fname ) + '' + >>> fname = get_test_fname('gff_version_3.gff') + >>> Gff3().sniff( fname ) + 'gff3' + """ + valid_gff3_strand = ['+', '-', '.', '?'] + valid_gff3_phase = ['.', '0', '1', '2'] + headers = get_headers( filename, '\t' ) + try: + if len(headers) < 2: + return '' + for hdr in headers: + if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '3' ) < 0: + return '' + if hdr and hdr[0] and not hdr[0].startswith( '#' ): + if len(hdr) != 9: + return '' + try: + map( int, [hdr[3]] ) + except: + if hdr[3] != '.': + return '' + try: + map( int, [hdr[4]] ) + except: + if hdr[4] != '.': + return '' + if hdr[5] != '.': + try: + score = int(hdr[5]) + except: + return '' + if (score < 0 or score > 1000): + return '' + if hdr[6] not in valid_gff3_strand: + return '' + if hdr[7] not in valid_gff3_phase: + return '' + return self.file_ext + except: + return '' + class Wiggle( Tabular ): """Tab delimited data in wiggle format""" + file_ext = "wig" + MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True ) def make_html_table(self, data): return Tabular.make_html_table(self, data, skipchar='#') + + def sniff( self, filename ): + """ + Determines wether the file is in wiggle format -#Extend Tabular type, since interval tools will fail on track def line (we should fix this) -#This is a skeleton class for now, allows viewing at ucsc and formatted peeking. + The .wig format is line-oriented. Wiggle data is preceeded by a track definition line, + which adds a number of options for controlling the default display of this track. + Following the track definition line is the track data, which can be entered in several + different formats. + + The track definition line begins with the word 'track' followed by the track type. + The track type with version is REQUIRED, and it currently must be wiggle_0. For example, + track type=wiggle_0... + + For complete details see http://genome.ucsc.edu/goldenPath/help/wiggle.html + + >>> fname = get_test_fname( 'interval.bed' ) + >>> Wiggle().sniff( fname ) + '' + >>> fname = get_test_fname( 'wiggle.wig' ) + >>> Wiggle().sniff( fname ) + 'wig' + """ + headers = get_headers( filename, None ) + try: + for hdr in headers: + if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'): + return self.file_ext + return '' + except: + return '' + class CustomTrack ( Tabular ): """UCSC CustomTrack""" - + file_ext = "customtrack" + def __init__(self, **kwd): """Initialize interval datatype, by adding UCSC display app""" Tabular.__init__(self, **kwd) @@ -394,9 +654,49 @@ class CustomTrack ( Tabular ): ret_val.append( (site_name, link) ) return ret_val -#Extend Tabular type, since interval tools will fail on track def line (we should fix this) -#This is a skeleton class for now, allows viewing at GBrowse and formatted peeking. + def sniff( self, filename ): + """ + Determines whether the file is in customtrack format. + + CustomTrack files are built within Galaxy and are basically bed or interval files with the first line looking + something like this. + + track name="User Track" description="User Supplied Track (from Galaxy)" color=0,0,0 visibility=1 + + >>> fname = get_test_fname( 'complete.bed' ) + >>> CustomTrack().sniff( fname ) + '' + >>> fname = get_test_fname( 'ucsc.customtrack' ) + >>> CustomTrack().sniff( fname ) + 'customtrack' + """ + headers = get_headers( filename, None ) + first_line = True + for hdr in headers: + if first_line: + try: + if hdr[0].startswith('track'): + first_line = False + else: + return '' + except: + return '' + else: + try: + if not (hdr[0] == '' or hdr[0].startswith( '#' )): + if len(hdr) < 3: + return '' + try: + map( int, [hdr[1], hdr[2]] ) + except: + return '' + except: + return '' + return self.file_ext + class GBrowseTrack ( Tabular ): + """GMOD GBrowseTrack""" + file_ext = "gbrowsetrack" def __init__(self, **kwd): """Initialize datatype, by adding GBrowse display app""" @@ -431,6 +731,15 @@ class GBrowseTrack ( Tabular ): #TODO: fix me... return open(dataset.file_name) + def sniff( self, filename ): + """ + Determines whether the file is in gbrowsetrack format. + + GBrowseTrack files are built within Galaxy. + TODO: Not yet sure what this file will look like. Fix this sniffer and add some unit tests here as soon as we know. + """ + return '' + if __name__ == '__main__': import doctest, sys doctest.testmod(sys.modules[__name__]) diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index 83ab5c5997c..ddbd9198d49 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -8,11 +8,12 @@ import galaxy.util from galaxy.util.odict import odict class Registry( object ): - def __init__( self, datatypes = [] ): + def __init__( self, datatypes=[], sniff_order=[] ): self.log = logging.getLogger(__name__) self.datatypes_by_extension = {} self.mimetypes_by_extension = {} self.datatype_converters = odict() + self.sniff_order = [] for ext, kind in datatypes: try: mime_type = None @@ -38,8 +39,7 @@ class Registry( object ): self.datatypes_by_extension = { 'data' : data.Data(), 'bed' : interval.Bed(), - 'txt' : data.Text(), - 'text' : data.Text(), + 'txt' : data.Text(), 'interval' : interval.Interval(), 'tabular' : tabular.Tabular(), 'png' : images.Image(), @@ -54,14 +54,12 @@ class Registry( object ): 'laj' : images.Laj(), 'lav' : sequence.Lav(), 'html' : images.Html(), - 'customtrack' : interval.CustomTrack(), - 'gbrowsetrack' : interval.GBrowseTrack() + 'customtrack' : interval.CustomTrack() } self.mimetypes_by_extension = { 'data' : 'application/octet-stream', 'bed' : 'text/plain', - 'txt' : 'text/plain', - 'text' : 'text/plain', + 'txt' : 'text/plain', 'interval' : 'text/plain', 'tabular' : 'text/plain', 'png' : 'image/png', @@ -76,14 +74,63 @@ class Registry( object ): 'laj' : 'text/plain', 'lav' : 'text/plain', 'html' : 'text/html', - 'customtrack' : 'text/plain', - 'gbrowsetrack' : 'text/plain' + 'customtrack' : 'text/plain' } + """ + The order in which we attempt to determine data types is critical + because some formats are much more flexibly defined than others. + """ + sniff_order.sort() + for ord, kind in sniff_order: + try: + fields = kind.split(":") + datatype_module = fields[0] + datatype_class = fields[1] + fields = datatype_module.split(".") + module = __import__(fields.pop(0)) + for mod in fields: module = getattr(module,mod) + aclass = getattr(module, datatype_class)() + included = False + for atype in self.sniff_order: + if isinstance(atype, aclass.__class__): + included = True + break + if not included: + self.sniff_order.append(aclass) + except Exception, exc: + self.log.warning('error appending datatype: %s to sniff_order, error: %s' %(str(kind), str(exc))) + #default values + if len(self.sniff_order) < 1: + self.sniff_order = [ + images.Gmaj(), + images.Laj(), + sequence.Maf(), + sequence.Lav(), + sequence.Fasta(), + interval.Wiggle(), + images.Html(), + sequence.Axt(), + interval.Bed(), + interval.CustomTrack(), + interval.Gff(), + interval.Gff3(), + interval.Interval() + ] + def append_to_sniff_order(): + """Just in case any supported data types are not included in the config's sniff_order section.""" + for ext in self.datatypes_by_extension: + datatype = self.datatypes_by_extension[ext] + included = False + for atype in self.sniff_order: + if isinstance(atype, datatype.__class__): + included = True + break + if not included: + self.sniff_order.append(datatype) + append_to_sniff_order() def get_mimetype_by_extension(self, ext ): - """ - Returns a mimetype based on an extension - """ + """Returns a mimetype based on an extension""" try: mimetype = self.mimetypes_by_extension[ext] except KeyError: @@ -93,9 +140,7 @@ class Registry( object ): return mimetype def get_datatype_by_extension(self, ext ): - """ - Returns a datatype based on an extension - """ + """Returns a datatype based on an extension""" try: builder = self.datatypes_by_extension[ext] except KeyError: @@ -114,9 +159,7 @@ class Registry( object ): return data def old_change_datatype(self, data, ext): - """ - Creates and returns a new datatype based on an existing data and an extension - """ + """Creates and returns a new datatype based on an existing data and an extension""" newdata = factory(ext)(id=data.id) for key, value in data.__dict__.items(): setattr(newdata, key, value) diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index cb07afcf407..63db034b0ef 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -7,6 +7,7 @@ import logging from galaxy.datatypes.metadata import MetadataElement from galaxy.datatypes import metadata from galaxy import util +from sniff import * log = logging.getLogger(__name__) @@ -22,6 +23,7 @@ class Alignment( Sequence ): class Fasta( Sequence ): """Class representing a FASTA sequence""" + file_ext = "fasta" def set_peek( self, dataset ): Sequence.set_peek( self, dataset ) @@ -37,14 +39,42 @@ class Fasta( Sequence ): else: dataset.blurb = '%d sequences' % count + def sniff(self, filename): + """ + Determines whether the file is in fasta format + + A sequence in FASTA format consists of a single-line description, followed by lines of sequence data. + The first character of the description line is a greater-than (">") symbol in the first column. + All lines should be shorter than 80 charcters + + For complete details see http://www.g2l.bio.uni-goettingen.de/blast/fastades.html + + >>> fname = get_test_fname( 'sequence.maf' ) + >>> Fasta().sniff( fname ) + '' + >>> fname = get_test_fname( 'sequence.fasta' ) + >>> Fasta().sniff( fname ) + 'fasta' + """ + headers = get_headers( filename, None ) + try: + if len(headers) > 1 and headers[0][0] and headers[0][0][0] == ">": + return self.file_ext + else: + return '' + except: + return '' + try: import pkg_resources; pkg_resources.require( "bx-python" ) import bx.align.maf except: pass + class Maf( Alignment ): """Class describing a Maf alignment""" - + file_ext = "maf" + def init_meta( self, dataset, copy_from=None ): Alignment.init_meta( self, dataset, copy_from=copy_from ) @@ -77,9 +107,106 @@ class Maf( Alignment ): return True return False + def sniff( self, filename ): + """ + Determines wether the file is in maf format + + The .maf format is line-oriented. Each multiple alignment ends with a blank line. + Each sequence in an alignment is on a single line, which can get quite long, but + there is no length limit. Words in a line are delimited by any white space. + Lines starting with # are considered to be comments. Lines starting with ## can + be ignored by most programs, but contain meta-data of one form or another. + + The first line of a .maf file begins with ##maf. This word is followed by white-space-separated + variable=value pairs. There should be no white space surrounding the "=". + + For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format5 + + >>> fname = get_test_fname( 'sequence.maf' ) + >>> Maf().sniff( fname ) + 'maf' + >>> fname = get_test_fname( 'sequence.fasta' ) + >>> Maf().sniff( fname ) + '' + """ + headers = get_headers( filename, None ) + try: + if len(headers) > 1 and headers[0][0] and headers[0][0] == "##maf": + return self.file_ext + else: + return '' + except: + return '' + class Axt( Alignment ): """Class describing an axt alignment""" + file_ext = "axt" + + def sniff( self, filename ): + """ + Determines whether the file is in axt format + + axt alignment files are produced from Blastz, an alignment tool available from Webb Miller's lab + at Penn State University. Each alignment block in an axt file contains three lines: a summary + line and 2 sequence lines. Blocks are separated from one another by blank lines. + + The summary line contains chromosomal position and size information about the alignment. It consists of 9 required fields: + + For complete details see http://genome.ucsc.edu/goldenPath/help/axt.html + + >>> fname = get_test_fname( 'alignment.axt' ) + >>> Axt().sniff( fname ) + 'axt' + >>> fname = get_test_fname( 'alignment.lav' ) + >>> Axt().sniff( fname ) + '' + """ + headers = get_headers( filename, None ) + if len(headers) < 4: + return '' + try: + """Assume the summary line is the first line of the file.""" + line = headers[0] + except: + return '' + + if len(line) != 9: + return '' + try: + map ( int, [line[0], line[2], line[3], line[5], line[6], line[8]] ) + except: + return '' + if line[7] not in data.valid_strand: + return '' + return self.file_ext class Lav( Alignment ): """Class describing a LAV alignment""" + file_ext = "lav" + + def sniff( self, filename ): + """ + Determines whether the file is in lav format + + LAV is an alignment format developed by Webb Miller's group. It is the primary output format for BLASTZ. + The first line of a .lav file begins with #:lav. + + For complete details see http://www.bioperl.org/wiki/LAV_alignment_format + + >>> fname = get_test_fname( 'alignment.lav' ) + >>> Lav().sniff( fname ) + 'lav' + >>> fname = get_test_fname( 'alignment.axt' ) + >>> Lav().sniff( fname ) + '' + """ + headers = get_headers( filename, None ) + try: + if len(headers) > 1 and headers[0][0] and headers[0][0].startswith('#:lav'): + return self.file_ext + else: + return '' + except: + return '' + diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index 9ecc72446fa..2f3a1e80a0e 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -2,12 +2,9 @@ File format detector """ import logging, sys, os, csv, tempfile, shutil, re +import registry log = logging.getLogger(__name__) - -valid_strand = ['+', '-', '.'] -valid_gff3_strand = ['+', '-', '.', '?'] -valid_gff3_phase = ['.', '0', '1', '2'] def get_test_fname(fname): """Returns test data filename""" @@ -115,416 +112,18 @@ def is_column_based(fname, sep='\t', skip=0): if not headers: return False - for hdr in headers[skip:]: if len(hdr) > 1 and hdr[0] and not hdr[0].startswith('#'): count = len(hdr) break - if count < 2: return False - for hdr in headers[skip:]: if len(hdr) > 1 and hdr[0] and not hdr[0].startswith('#') and len(hdr) != count: return False return True - -def is_fasta(headers): - """ - Determines wether the file is in fasta format - - A sequence in FASTA format consists of a single-line description, followed by lines of sequence data. - The first character of the description line is a greater-than (">") symbol in the first column. - All lines should be shorter than 80 charcters - - For complete details see http://www.g2l.bio.uni-goettingen.de/blast/fastades.html - - >>> headers = get_headers(__file__, ' ') - >>> is_fasta(headers) - False - >>> fname = get_test_fname('sequence.fasta') - >>> headers = get_headers(fname,' ') - >>> is_fasta(headers) - True - """ - try: - return len(headers) > 1 and headers[0][0] and headers[0][0][0] == ">" - except: - return False -def is_gff(headers): - """ - Determines wether the file is in gff format - - GFF lines have nine required fields that must be tab-separated. - - For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format3 - - >>> headers = get_headers(__file__,' ') - >>> is_fasta(headers) - False - >>> fname = get_test_fname('test.gff') - >>> headers = get_headers(fname,'\\t') - >>> is_gff(headers) - True - """ - try: - if len(headers) < 2: - return False - for hdr in headers: - if len( hdr ) > 1 and hdr[0] and not hdr[0].startswith( '#' ): - if len(hdr) != 9: - return False - try: - map( int, [hdr[3], hdr[4]] ) - except: - return False - if hdr[5] != '.': - try: - score = int(hdr[5]) - except: - return False - if (score < 0 or score > 1000): - return False - if hdr[6] not in valid_strand: - return False - return True - except: - return False - -def is_gff3(headers): - """ - Determines wether the file is in gff version 3 format - - GFF 3 format: - - 1) adds a mechanism for representing more than one level - of hierarchical grouping of features and subfeatures. - 2) separates the ideas of group membership and feature name/id - 3) constrains the feature type field to be taken from a controlled - vocabulary. - 4) allows a single feature, such as an exon, to belong to more than - one group at a time. - 5) provides an explicit convention for pairwise alignments - 6) provides an explicit convention for features that occupy disjunct regions - - The format consists of 9 columns, separated by tabs (NOT spaces). - - Undefined fields are replaced with the "." character, as described in the original GFF spec. - - For complete details see http://song.sourceforge.net/gff3.shtml - - >>> headers = get_headers(__file__,' ') - >>> is_fasta(headers) - False - >>> fname = get_test_fname('gff_version_3.gff') - >>> headers = get_headers(fname,'\\t') - >>> is_gff3(headers) - True - """ - try: - if len(headers) < 2: - return False - for hdr in headers: - if len(hdr) > 1 and hdr[0] and not hdr[0].startswith( '#' ): - if len(hdr) != 9: - return False - try: - map( int, [hdr[3]] ) - except: - if hdr[3] != '.': - return False - try: - map( int, [hdr[4]] ) - except: - if hdr[4] != '.': - return False - if hdr[5] != '.': - try: - score = int(hdr[5]) - except: - return False - if (score < 0 or score > 1000): - return False - if hdr[6] not in valid_gff3_strand: - return False - if hdr[7] not in valid_gff3_phase: - return False - return True - except: - return False - -def is_maf(headers): - """ - Determines wether the file is in maf format - - The .maf format is line-oriented. Each multiple alignment ends with a blank line. - Each sequence in an alignment is on a single line, which can get quite long, but - there is no length limit. Words in a line are delimited by any white space. - Lines starting with # are considered to be comments. Lines starting with ## can - be ignored by most programs, but contain meta-data of one form or another. - - The first line of a .maf file begins with ##maf. This word is followed by white-space-separated - variable=value pairs. There should be no white space surrounding the "=". - - For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format5 - - >>> headers = get_headers(__file__,' ') - >>> is_maf(headers) - False - >>> fname = get_test_fname('sequence.maf') - >>> headers = get_headers(fname,' ') - >>> is_maf(headers) - True - """ - try: - return len(headers) > 1 and headers[0][0] and headers[0][0] == "##maf" - except: - return False - -def is_lav(headers): - """ - Determines wether the file is in lav format - - LAV is an alignment format developed by Webb Miller's group. It is the primary output format for BLASTZ. - The first line of a .lav file begins with #:lav. - - For complete details see http://www.bioperl.org/wiki/LAV_alignment_format - - >>> headers = get_headers(__file__,' ') - >>> is_lav(headers) - False - >>> fname = get_test_fname('alignment.lav') - >>> headers = get_headers(fname,' ') - >>> is_lav(headers) - True - """ - try: - return len(headers) > 1 and headers[0][0] and headers[0][0].startswith('#:lav') - except: - return False - -def is_axt(headers): - """ - Determines wether the file is in axt format - - axt alignment files are produced from Blastz, an alignment tool available from Webb Miller's lab - at Penn State University. Each alignment block in an axt file contains three lines: a summary - line and 2 sequence lines. Blocks are separated from one another by blank lines. - - The summary line contains chromosomal position and size information about the alignment. It consists of 9 required fields: - - For complete details see http://genome.ucsc.edu/goldenPath/help/axt.html - - >>> headers = get_headers(__file__,None) - >>> is_axt(headers) - False - >>> fname = get_test_fname('alignment.axt') - >>> headers = get_headers(fname,None) - >>> is_axt(headers) - True - """ - try: - #return (len(headers) >= 4) and (headers[0][7] == '-' or headers[0][7] == '+') and (headers[3] == []) and (len(headers[0])==9 or len(headers[0])==10) - if len(headers) < 4: - return False - """ - Assume the summary line is the first line of the file. - """ - line = headers[0] - - if len(line) != 9: - return False - try: - map ( int, [line[0], line[2], line[3], line[5], line[6], line[8]] ) - except: - return False - if line[7] not in valid_strand: - return False - return True - except: - return False - -def is_wiggle(headers): - """ - Determines wether the file is in wiggle format - - The .wig format is line-oriented. Wiggle data is preceeded by a track definition line, - which adds a number of options for controlling the default display of this track. - Following the track definition line is the track data, which can be entered in several - different formats. - - The track definition line begins with the word 'track' followed by the track type. - The track type with version is REQUIRED, and it currently must be wiggle_0. For example, - track type=wiggle_0... - - For complete details see http://genome.ucsc.edu/goldenPath/help/wiggle.html - - >>> headers = get_headers(__file__,' ') - >>> is_wiggle(headers) - False - >>> fname = get_test_fname('wiggle.wig') - >>> headers = get_headers(fname,' ') - >>> is_wiggle(headers) - True - """ - try: - for hdr in headers: - if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'): - return True - return False - except: - return False - -def is_bed(headers): - """ - Checks for 'bedness' - - BED lines have three required fields and nine additional optional fields. - The number of fields per line must be consistent throughout any single set of data in - an annotation track. - - For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format1 - - >>> fname = get_test_fname('test_tab.bed') - >>> headers = get_headers(fname,'\\t') - >>> is_bed(headers) - True - >>> fname = get_test_fname('interval.bed') - >>> headers = get_headers(fname,'\\t') - >>> is_bed(headers) - False - >>> fname = get_test_fname('complete.bed') - >>> headers = get_headers(fname,'\\t') - >>> is_bed(headers) - True - """ - - col1_startswith = ['chr', 'chl', 'groupun', 'reftig_', 'scaffold', 'super_', 'vcho'] - - try: - if not headers: - return False - for hdr in headers: - valid_col1 = False - if len(hdr) < 3 or len(hdr) > 12: - return False - for str in col1_startswith: - if hdr[0].lower().startswith(str): - valid_col1 = True - break - if valid_col1: - try: - map( int, [hdr[1], hdr[2]] ) - except: - return False - if len(hdr) > 3: - """ - Since all 9 of these fields are optional, it is difficult to test - for specific column values... - """ - optionals = hdr[3:] - """ - ...we can, however, test complete BED definitions fairly easily. - """ - if len(optionals) == 9: - try: - map ( int, [optionals[1], optionals[3], optionals[4], optionals[5], optionals[6]] ) - except: - return False - score = int(optionals[1]) - if score < 0 or score > 1000: - return False - if optionals[2] not in ['+', '-']: - return False - if int(optionals[5]) != 0: - return False - block_count = int(optionals[6]) - """ - Sometime the blosck_sizes and block_starts lists end in extra commas - """ - block_sizes = optionals[7].rstrip(',').split(',') - block_starts = optionals[8].rstrip(',').split(',') - if len(block_sizes) != block_count or len(block_starts) != block_count: - return False - elif len(optionals) > 4 and len(optionals) < 9: - """ - Here it gets a bit trickier, but in this case, we can be somewhat confident - that optionals will include a strand column - """ - is_valid_strand = False - for ele in optionals: - if ele in valid_bed_strand: - is_valid_strand = True - if not is_valid_strand: - return False - else: - return False - return True - except: - return False - -def is_interval(headers): - """ - Checks for 'intervalness' - - This is the most loosely defined format and is mostly used by galaxy itself. In general, if - the format is_column_based, but not any of the other formats, then it must be interval. - - >>> fname = get_test_fname('test_space.bed') - >>> headers = get_headers(fname,'\\t') - >>> is_interval(headers) - False - >>> fname = get_test_fname('interval.interval') - >>> headers = get_headers(fname,'\\t') - >>> is_interval(headers) - True - """ - - try: - #return is_bed(headers, skip=1) and headers[0][0][0] == '#' - """ - If we got here, we already know the file is_column_based and is not bed, - so we'll just look for some valid data. - """ - for hdr in headers: - if not (hdr[0] == '' or hdr[0].startswith( '#' )): - if len(hdr) < 3: - return False - try: - map( int, [hdr[1], hdr[2]] ) - except: - return False - return True - except: - return False - -def is_html(headers): - """ - Determines wether the file is in html format - - >>> headers = get_headers(__file__,' ') - >>> is_html(headers) - False - >>> fname = get_test_fname('file.html') - >>> headers = get_headers(fname,' ') - >>> is_html(headers) - True - """ - try: - for idx, hdr in enumerate(headers): - if hdr and hdr[0].lower() == '': - return True - if idx > 29: - """ - This is a weakness since it assumes < 29 blank lines, comments, etc. - """ - break - return False - except: - return False - -def guess_ext(fname): +def guess_ext( fname ): """ Returns an extension that can be used in the datatype factory to generate a data for the 'fname' file @@ -552,65 +151,46 @@ def guess_ext(fname): 'gff' >>> fname = get_test_fname('gff_version_3.gff') >>> guess_ext(fname) - 'gff' + 'gff3' >>> fname = get_test_fname('temp.txt') - >>> file(fname, 'wt').write("a 2\\nc 1") + >>> file(fname, 'wt').write("a 2\\nc 1\\nd 0") >>> guess_ext(fname) 'tabular' >>> fname = get_test_fname('temp.txt') - >>> file(fname, 'wt').write("a 1 2 x\\nb 3 4 y") + >>> file(fname, 'wt').write("a 1 2 x\\nb 3 4 y\\nc 5 6 z") >>> guess_ext(fname) - 'interval' - - """ - try: + 'tabular' + """ + datatypes_registry = registry.Registry() + for datatype in datatypes_registry.sniff_order: """ - The order in which we attempt to determine data format is pretty important - because some formats are much more flexibly defined than others. Interval format - is the most loosely defined, so it should be that last format we check. + Some classes may not have a sniff function, which is ok. In fact, the + Tabular and Text classes are 2 examples of classes that should never have + a sniff function. Since these classes are default classes, they contain + few rules to filter out data of other formats, so they should be called + from this function after all other datatypes in sniff_order have not been + successfully discovered. """ - #guess if data is binary - for line in file(fname): - for char in line: - if ord(char) > 128: - return "data" - - headers = get_headers(fname, None) - if is_maf(headers): - return 'maf' - elif is_lav(headers): - return 'lav' - elif is_fasta(headers): - return 'fasta' - elif is_wiggle(headers): - return 'wig' - elif is_html(headers): - return 'html' - elif is_axt(headers): - return 'axt' + try: + format = datatype.sniff( fname ) + except: + format = '' + if format: + return format - if is_column_based(fname, ' ', 1): - sep2tabs(fname) - - if is_column_based(fname, '\t', 0): - headers = get_headers(fname, '\t') - if is_bed(headers): - return 'bed' - - if is_column_based(fname, '\t', 1): - headers = get_headers(fname, '\t') - if is_gff(headers): - return 'gff' - if is_gff3(headers): - return 'gff3' - elif is_interval(headers): - return 'interval' + """Default binary file extension""" + for line in file( fname ): + for char in line: + if ord(char) > 128: + return "data" else: - return 'tabular' - except: - pass - - return 'text' + break + break + if is_column_based( fname, ' ', 1 ): + sep2tabs(fname) + if is_column_based( fname, '\t', 1): + return "tabular" + return 'txt' if __name__ == '__main__': import doctest, sys diff --git a/lib/galaxy/datatypes/tabular.py b/lib/galaxy/datatypes/tabular.py index 0d7ac0e17c4..ca4a8da7c99 100644 --- a/lib/galaxy/datatypes/tabular.py +++ b/lib/galaxy/datatypes/tabular.py @@ -6,7 +6,7 @@ import pkg_resources pkg_resources.require( "bx-python" ) import logging -import data, sniff +import data from galaxy import util from cgi import escape from galaxy.datatypes import metadata @@ -15,7 +15,6 @@ from galaxy.datatypes.metadata import MetadataAttributes log = logging.getLogger(__name__) - class Tabular( data.Text ): """Tab delimited data""" @@ -40,118 +39,47 @@ class Tabular( data.Text ): """ if dataset.has_data(): column_types = [] - format = sniff.guess_ext( dataset.file_name ) - - if dataset.extension != format: - """ - We need to rely on the ability of our sniffer to properly detect datatypes here since - there are many ways that the datatype could be improperly set. - TODO: we may want to automatically convert the datset to the proper datatype here, - but for now we'll leave it and just rely on the value in format. - """ - pass - - """ - The 'proceed' value allows us to skip different numbers of lines based on the data - format (type). We need this because different formats can include lines of information - that are not properly commented (properly commented lines start with a '#' character). - """ - proceed = False - col1_startswith = ['chr', 'chl', 'groupun', 'reftig_', 'scaffold', 'super_', 'vcho'] for i, line in enumerate( file ( dataset.file_name ) ): line = line.rstrip('\r\n') - valid = True if line and not line.startswith( '#' ): elems = line.split( '\t' ) elems_len = len(elems) if elems_len > 0: - if format == 'bed': - for str in col1_startswith: - if elems[0].lower().startswith(str): - proceed = True - break - elif format == 'interval': - if elems_len > 2: + """Set the columns metadata attribute""" + if elems_len != dataset.metadata.columns: + dataset.metadata.columns = elems_len + + """Set the column_types metadata attribute""" + for col in range(0, elems_len): + col_type = None + val = elems[col] + if not col_type: + """See if val is an int""" try: - map( int, [elems[1], elems[2]] ) - proceed = True - except: - pass # proceed is False - elif format == 'gff': - if elems_len == 9: - try: - map( int, [hdr[3], hdr[4]] ) - proceed = True - except: + int( val ) + col_type = 'int' + except: pass - elif format == 'gff3': - valid_gff3_strand = ['+', '-', '.', '?'] - valid_start = False - valid_end = False - if elems_len == 9: + if not col_type: + """See if val is a float""" try: - start = int(hdr[3]) - valid_start = True + float( val ) + col_type = 'float' except: - if hdr[3] == '.': - valid_start = True - try: - end = int(hdr[4]) - valid_end = True - except: - if hdr[4] == '.': - valid_end = True - srand = hdr[6] - if valid_start and valid_end and start < end and strand in valid_gff3_strand: - proceed = True - elif format=='wig': - try: - int( elems[0] ) - proceed = True - except: - for str in col1_startswith: - if elems[0].lower().startswith(str): - proceed = True - break - elif ( format =='tabular' or format == "customtrack" or format == 'gbrowsetrack' ) and i > 1: - proceed = True - - if proceed: - """Set the columns metadata attribute""" - if elems_len != dataset.metadata.columns: - dataset.metadata.columns = elems_len - - """Set the column_types metadata attribute""" - for col in range(0, elems_len): - col_type = None - val = elems[col] - if not col_type: - """See if val is an int""" - try: - int( val ) - col_type = 'int' - except: - pass - if not col_type: - """See if val is a float""" - try: - float( val ) + if val and val.strip().lower() == 'na': col_type = 'float' - except: - if val and val.strip().lower() == 'na': - col_type = 'float' - if not col_type: - """See if val is a list""" - val_elems = val.split(',') - if len( val_elems ) > 1: - col_type = 'list' - if not col_type: - """All parameters are strings, so this will be the default""" - col_type = 'str' + if not col_type: + """See if val is a list""" + val_elems = val.split(',') + if len( val_elems ) > 1: + col_type = 'list' + if not col_type: + """All parameters are strings, so this will be the default""" + col_type = 'str' - column_types.append(col_type) + column_types.append(col_type) if len(column_types) > 0: break if i > 100: break # Hopefully we never get here... @@ -204,4 +132,5 @@ class Tabular( data.Text ): def before_edit( self, dataset ): data.Text.before_edit( self, dataset ) if self.missing_meta( dataset ): - self.set_meta( dataset ) + self.set_meta( dataset ) + diff --git a/lib/galaxy/datatypes/test/test.gff b/lib/galaxy/datatypes/test/test.gff index eb2c1d1ff3e..997a2114b57 100644 --- a/lib/galaxy/datatypes/test/test.gff +++ b/lib/galaxy/datatypes/test/test.gff @@ -1,7 +1,7 @@ -# gff-version 2 -## Date: Thu Dec 8 19:46:27 2005 -## gff.pl $Rev: 601 $ -## Input file: /depot/data2/galaxy/encode-data/datasets/msa.AR.20051208.bed +##gff-version 2 +##Date: Thu Dec 8 19:46:27 2005 +##gff.pl $Rev: 601 $ +##Input file: /depot/data2/galaxy/encode-data/datasets/msa.AR.20051208.bed chr7 bed2gff AR 26731313 26731437 . + . score chr7 bed2gff AR 26731491 26731536 . + . score diff --git a/universe_wsgi.ini.sample b/universe_wsgi.ini.sample index ca2373b91f2..6fe6aa43e26 100644 --- a/universe_wsgi.ini.sample +++ b/universe_wsgi.ini.sample @@ -105,7 +105,6 @@ static_style_dir = %(here)s/static/june_2007_style/blue data = galaxy.datatypes.data:Data,application/octet-stream bed = galaxy.datatypes.interval:Bed txt = galaxy.datatypes.data:Text -text = galaxy.datatypes.data:Text interval = galaxy.datatypes.interval:Interval tabular = galaxy.datatypes.tabular:Tabular png = galaxy.datatypes.images:Image,image/png @@ -123,7 +122,6 @@ laj = galaxy.datatypes.images:Laj lav = galaxy.datatypes.sequence:Lav html = galaxy.datatypes.images:Html,text/html customtrack = galaxy.datatypes.interval:CustomTrack -gbrowsetrack = galaxy.datatypes.interval:GBrowseTrack #EMBOSS TOOLS match = galaxy.datatypes.data:Text genbank = galaxy.datatypes.data:Text @@ -171,3 +169,21 @@ swiss = galaxy.datatypes.data:Text phylipnon = galaxy.datatypes.data:Text nexusnon = galaxy.datatypes.data:Text nametable = galaxy.datatypes.data:Text + +# ---- Data Type Sniff Order -------------------------------------------------- + +[galaxy:sniff_order] + +05 = galaxy.datatypes.images:Gmaj +10 = galaxy.datatypes.images:Laj +15 = galaxy.datatypes.sequence:Maf +20 = galaxy.datatypes.sequence:Lav +25 = galaxy.datatypes.sequence:Fasta +30 = galaxy.datatypes.interval:Wiggle +35 = galaxy.datatypes.images:Html +40 = galaxy.datatypes.sequence:Axt +45 = galaxy.datatypes.interval:Bed +50 = galaxy.datatypes.interval:CustomTrack +55 = galaxy.datatypes.interval:Gff +60 = galaxy.datatypes.interval:Gff3 +65 = galaxy.datatypes.interval:Interval