Streamlined interface for supporting new data types - requires db change and note config change.

db change:
update dataset set extension = 'txt' where extension = 'text';

Adding support for a new data type is now as simple as adding the new type to the config file and optionally adding the
type into the new sniff_order section in the config file, and adding the new class for the data type to the appropriate
datatypes code.

The config file now has a new section [galaxy:sniff_order] which lists the order in which datatype class sniff functions will
be called (each datatype class can now optionally include its own sniff function, although not a requirement).

Support for the 'text' file extension has been eliminated as we will standardize on the 'txt' file extension for the Text data type.
This commit is contained in:
Greg Von Kuster
2007-09-27 14:10:16 +00:00
parent 0cc15f4a51
commit 1a6daca3a4
11 changed files with 644 additions and 598 deletions
+1 -1
View File
@@ -14,7 +14,7 @@ class UniverseApplication( object ):
self.config.check()
config.configure_logging( self.config )
#Set up datatypes registry
self.datatypes_registry = galaxy.datatypes.registry.Registry(datatypes = self.config.datatypes)
self.datatypes_registry = galaxy.datatypes.registry.Registry(datatypes=self.config.datatypes, sniff_order=self.config.sniff_order)
galaxy.model.set_datatypes_registry(self.datatypes_registry)
# Connect up the object model
if self.config.database_connection:
+5
View File
@@ -60,6 +60,11 @@ class Configuration( object ):
self.datatypes = global_conf_parser.items("galaxy:datatypes")
except ConfigParser.NoSectionError:
self.datatypes = []
#Store sniff order config
try:
self.sniff_order = global_conf_parser.items("galaxy:sniff_order")
except ConfigParser.NoSectionError:
self.sniff_order = []
self.datatype_converters_config = kwargs.get( 'datatype_converters_config_file', "datatype_converters_conf.xml" )
self.datatype_converters_path = kwargs.get( 'datatype_converters_path', os.path.join(self.root,"lib/galaxy/datatypes/converters") )
def get( self, key, default ):
+5 -1
View File
@@ -2,10 +2,14 @@ import logging, os, sys, time, sets, tempfile
from galaxy import util
from cgi import escape
from galaxy.datatypes.metadata import *
log = logging.getLogger(__name__)
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes import metadata
log = logging.getLogger(__name__)
# Valid strand column values
valid_strand = ['+', '-', '.']
# Constants for data states
DATA_NEW, DATA_OK, DATA_FAKE = 'new', 'ok', 'fake'
+44 -11
View File
@@ -4,6 +4,7 @@ Image classes
import data
import logging
from galaxy.datatypes.sniff import *
from urllib import urlencode
log = logging.getLogger(__name__)
@@ -17,6 +18,8 @@ class Image( data.Data ):
class Gmaj( data.Data ):
"""Class describing a GMAJ Applet"""
file_ext = "gmaj.zip"
def set_peek( self, dataset ):
dataset.peek = "<p align=\"center\"><applet code=\"edu.psu.bx.gmaj.MajApplet.class\" archive=\"/static/gmaj/gmaj.jar\" width=\"200\" height=\"30\" align=\"middle\"> <param name=bundle value=\"display?id="+str(dataset.id)+"&tofile=yes&toext=.zip\"> <param name=buttonlabel value=\"Launch GMAJ\"><param name=nobutton value=\"false\"><param name=urlpause value=\"100\"><param name=debug value=\"false\"><i>Your browser is not responding to the &lt;applet&gt; tag.</i></applet></p>"
dataset.blurb = 'GMAJ Multiple Alignment Viewer'
@@ -29,27 +32,57 @@ class Gmaj( data.Data ):
def get_mime(self):
"""Returns the mime type of the datatype"""
return 'application/zip'
def sniff( self, filename ):
#TODO: fix me
return ''
class Html( data.Text ):
"""Class describing an html file"""
file_ext = "html"
def set_peek( self, dataset ):
dataset.peek = "HTML file (%s)" % ( data.nice_size( dataset.get_size() ) )
dataset.blurb = data.nice_size( dataset.get_size() )
def get_mime(self):
"""Returns the mime type of the datatype"""
return 'text/html'
def sniff( self, filename ):
"""
Determines wether the file is in html format
>>> fname = get_test_fname( 'complete.bed' )
>>> Html().sniff( fname )
''
>>> fname = get_test_fname( 'file.html' )
>>> Html().sniff( fname )
'html'
"""
headers = get_headers( filename, None )
try:
for i, hdr in enumerate(headers):
if hdr and hdr[0].lower().find( '<html>' ) >=0:
return self.file_ext
return ''
except:
return ''
class Laj( data.Text ):
"""Class describing a LAJ Applet"""
file_ext = "laj"
def set_peek( self, dataset ):
export_url = "/history_add_to?"+urlencode({'history_id':dataset.history_id,'ext':'lav','name':'LAJ Output','info':'Added by LAJ','dbkey':dataset.dbkey})
dataset.peek = "<p align=\"center\"><applet code=\"edu.psu.cse.bio.laj.LajApplet.class\" archive=\"static/laj/laj.jar\" width=\"200\" height=\"30\"><param name=buttonlabel value=\"Launch LAJ\"><param name=title value=\"LAJ in Galaxy\"><param name=posturl value=\""+export_url+"\"><param name=alignfile1 value=\"display?id="+str(dataset.id)+"\"><param name=noseq value=\"true\"></applet></p>"
dataset.blurb = 'LAJ Multiple Alignment Viewer'
def display_peek(self, dataset):
try:
return dataset.peek
except:
return "peek unavailable"
class Html( data.Text ):
"""Class describing an html file"""
def set_peek( self, dataset ):
dataset.peek = "HTML file (%s)" % ( data.nice_size( dataset.get_size() ) )
dataset.blurb = data.nice_size( dataset.get_size() )
def get_mime(self):
"""Returns the mime type of the datatype"""
return 'text/html'
def sniff( self, filename ):
#TODO: fix me...
return ''
+315 -6
View File
@@ -8,6 +8,7 @@ pkg_resources.require( "bx-python" )
import logging, os, sys, time, sets, tempfile, shutil
import data
from galaxy import util
from galaxy.datatypes.sniff import *
from cgi import escape
import urllib
from bx.intervals.io import *
@@ -36,6 +37,7 @@ for key, value in alias_spec.items():
class Interval( Tabular ):
"""Tab delimited data containing interval information"""
file_ext = "interval"
"""Add metadata elements"""
MetadataElement( name="chromCol", desc="Chrom column", param=metadata.ColumnParameter )
@@ -44,7 +46,6 @@ class Interval( Tabular ):
MetadataElement( name="strandCol", desc="Strand column (click box & select)", param=metadata.ColumnParameter, optional=True, no_value=0 )
MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True, visible=False )
def __init__(self, **kwd):
"""Initialize interval datatype, by adding UCSC display apps"""
Tabular.__init__(self, **kwd)
@@ -179,8 +180,41 @@ class Interval( Tabular ):
"""Return options for removing errors along with a description"""
return [("lines","Remove erroneous lines")]
def sniff( self, filename ):
"""
Checks for 'intervalness'
This format is mostly used by galaxy itself. Valid interval files should include
a valid header comment, but this seems to be loosely regulated.
>>> fname = get_test_fname( 'test_space.bed' )
>>> Interval().sniff( fname )
''
>>> fname = get_test_fname( 'interval.interval' )
>>> Interval().sniff( fname )
'interval'
"""
headers = get_headers( filename, '\t' )
try:
"""
If we got here, we already know the file is_column_based and is not bed,
so we'll just look for some valid data.
"""
for hdr in headers:
if not (hdr[0] == '' or hdr[0].startswith( '#' )):
if len(hdr) < 3:
return ''
try:
map( int, [hdr[1], hdr[2]] )
except:
return ''
return self.file_ext
except:
return ''
class Bed( Interval ):
"""Tab delimited data in BED format"""
file_ext = "bed"
"""Add metadata elements"""
MetadataElement( name="chromCol", default=1, desc="Chrom column", param=metadata.ColumnParameter )
@@ -264,8 +298,93 @@ class Bed( Interval ):
try: return open(dataset.file_name)
except: return "This item contains no content"
def sniff( self, filename ):
"""
Checks for 'bedness'
BED lines have three required fields and nine additional optional fields.
The number of fields per line must be consistent throughout any single set of data in
an annotation track.
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format1
>>> fname = get_test_fname( 'test_tab.bed' )
>>> Bed().sniff( fname )
'bed'
>>> fname = get_test_fname( 'interval.bed' )
>>> Bed().sniff( fname )
''
>>> fname = get_test_fname( 'complete.bed' )
>>> Bed().sniff( fname )
'bed'
"""
col1_startswith = ['chr', 'chl', 'groupun', 'reftig_', 'scaffold', 'super_', 'vcho']
headers = get_headers( filename, '\t' )
try:
if not headers:
return ''
for hdr in headers:
valid_col1 = False
if len(hdr) < 3 or len(hdr) > 12:
return ''
for str in col1_startswith:
if hdr[0].lower().startswith(str):
valid_col1 = True
break
if valid_col1:
try:
map( int, [hdr[1], hdr[2]] )
except:
return ''
if len(hdr) > 3:
"""
Since all 9 of these fields are optional, it is difficult to test
for specific column values...
"""
optionals = hdr[3:]
"""
...we can, however, test complete BED definitions fairly easily.
"""
if len(optionals) == 9:
try:
map ( int, [optionals[1], optionals[3], optionals[4], optionals[5], optionals[6]] )
except:
return ''
score = int(optionals[1])
if score < 0 or score > 1000:
return ''
if optionals[2] not in ['+', '-']:
return ''
if int(optionals[5]) != 0:
return ''
block_count = int(optionals[6])
"""
Sometimes the blosck_sizes and block_starts lists end in extra commas
"""
block_sizes = optionals[7].rstrip(',').split(',')
block_starts = optionals[8].rstrip(',').split(',')
if len(block_sizes) != block_count or len(block_starts) != block_count:
return ''
elif len(optionals) > 4 and len(optionals) < 9:
"""
Here it gets a bit trickier, but in this case, we can be somewhat confident
that optionals will include a strand column
"""
is_valid_strand = False
for ele in optionals:
if ele in data.valid_strand:
is_valid_strand = True
if not is_valid_strand:
return ''
else:
return ''
return self.file_ext
except:
return ''
class Gff( Tabular ):
"""Tab delimited data in Gff format"""
file_ext = "gff"
"""Add metadata elements"""
MetadataElement( name="columns", default=9, desc="Number of columns", readonly=True )
@@ -330,8 +449,51 @@ class Gff( Tabular ):
ret_val.append( (site_name, link) )
return ret_val
def sniff( self, filename ):
"""
Determines whether the file is in gff format
GFF lines have nine required fields that must be tab-separated.
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format3
>>> fname = get_test_fname( 'gff_version_3.gff' )
>>> Gff().sniff( fname )
''
>>> fname = get_test_fname( 'test.gff' )
>>> Gff().sniff( fname )
'gff'
"""
headers = get_headers( filename, '\t' )
try:
if len(headers) < 2:
return ''
for hdr in headers:
if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '2' ) < 0:
return ''
if hdr and hdr[0] and not hdr[0].startswith( '#' ):
if len(hdr) != 9:
return ''
try:
map( int, [hdr[3], hdr[4]] )
except:
return ''
if hdr[5] != '.':
try:
score = int(hdr[5])
except:
return ''
if (score < 0 or score > 1000):
return ''
if hdr[6] not in data.valid_strand:
return ''
return self.file_ext
except:
return ''
class Gff3( Gff ):
"""Tab delimited data in Gff3 format"""
file_ext = "gff3"
"""Add metadata elements"""
MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], desc="Column types", readonly=True )
@@ -340,18 +502,116 @@ class Gff3( Gff ):
"""Initialize datatype, by adding GBrowse display app"""
Gff.__init__(self, **kwd)
def sniff( self, filename ):
"""
Determines whether the file is in gff version 3 format
GFF 3 format:
1) adds a mechanism for representing more than one level
of hierarchical grouping of features and subfeatures.
2) separates the ideas of group membership and feature name/id
3) constrains the feature type field to be taken from a controlled
vocabulary.
4) allows a single feature, such as an exon, to belong to more than
one group at a time.
5) provides an explicit convention for pairwise alignments
6) provides an explicit convention for features that occupy disjunct regions
The format consists of 9 columns, separated by tabs (NOT spaces).
Undefined fields are replaced with the "." character, as described in the original GFF spec.
For complete details see http://song.sourceforge.net/gff3.shtml
>>> fname = get_test_fname( 'test.gff' )
>>> Gff3().sniff( fname )
''
>>> fname = get_test_fname('gff_version_3.gff')
>>> Gff3().sniff( fname )
'gff3'
"""
valid_gff3_strand = ['+', '-', '.', '?']
valid_gff3_phase = ['.', '0', '1', '2']
headers = get_headers( filename, '\t' )
try:
if len(headers) < 2:
return ''
for hdr in headers:
if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '3' ) < 0:
return ''
if hdr and hdr[0] and not hdr[0].startswith( '#' ):
if len(hdr) != 9:
return ''
try:
map( int, [hdr[3]] )
except:
if hdr[3] != '.':
return ''
try:
map( int, [hdr[4]] )
except:
if hdr[4] != '.':
return ''
if hdr[5] != '.':
try:
score = int(hdr[5])
except:
return ''
if (score < 0 or score > 1000):
return ''
if hdr[6] not in valid_gff3_strand:
return ''
if hdr[7] not in valid_gff3_phase:
return ''
return self.file_ext
except:
return ''
class Wiggle( Tabular ):
"""Tab delimited data in wiggle format"""
file_ext = "wig"
MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True )
def make_html_table(self, data):
return Tabular.make_html_table(self, data, skipchar='#')
def sniff( self, filename ):
"""
Determines wether the file is in wiggle format
#Extend Tabular type, since interval tools will fail on track def line (we should fix this)
#This is a skeleton class for now, allows viewing at ucsc and formatted peeking.
The .wig format is line-oriented. Wiggle data is preceeded by a track definition line,
which adds a number of options for controlling the default display of this track.
Following the track definition line is the track data, which can be entered in several
different formats.
The track definition line begins with the word 'track' followed by the track type.
The track type with version is REQUIRED, and it currently must be wiggle_0. For example,
track type=wiggle_0...
For complete details see http://genome.ucsc.edu/goldenPath/help/wiggle.html
>>> fname = get_test_fname( 'interval.bed' )
>>> Wiggle().sniff( fname )
''
>>> fname = get_test_fname( 'wiggle.wig' )
>>> Wiggle().sniff( fname )
'wig'
"""
headers = get_headers( filename, None )
try:
for hdr in headers:
if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'):
return self.file_ext
return ''
except:
return ''
class CustomTrack ( Tabular ):
"""UCSC CustomTrack"""
file_ext = "customtrack"
def __init__(self, **kwd):
"""Initialize interval datatype, by adding UCSC display app"""
Tabular.__init__(self, **kwd)
@@ -394,9 +654,49 @@ class CustomTrack ( Tabular ):
ret_val.append( (site_name, link) )
return ret_val
#Extend Tabular type, since interval tools will fail on track def line (we should fix this)
#This is a skeleton class for now, allows viewing at GBrowse and formatted peeking.
def sniff( self, filename ):
"""
Determines whether the file is in customtrack format.
CustomTrack files are built within Galaxy and are basically bed or interval files with the first line looking
something like this.
track name="User Track" description="User Supplied Track (from Galaxy)" color=0,0,0 visibility=1
>>> fname = get_test_fname( 'complete.bed' )
>>> CustomTrack().sniff( fname )
''
>>> fname = get_test_fname( 'ucsc.customtrack' )
>>> CustomTrack().sniff( fname )
'customtrack'
"""
headers = get_headers( filename, None )
first_line = True
for hdr in headers:
if first_line:
try:
if hdr[0].startswith('track'):
first_line = False
else:
return ''
except:
return ''
else:
try:
if not (hdr[0] == '' or hdr[0].startswith( '#' )):
if len(hdr) < 3:
return ''
try:
map( int, [hdr[1], hdr[2]] )
except:
return ''
except:
return ''
return self.file_ext
class GBrowseTrack ( Tabular ):
"""GMOD GBrowseTrack"""
file_ext = "gbrowsetrack"
def __init__(self, **kwd):
"""Initialize datatype, by adding GBrowse display app"""
@@ -431,6 +731,15 @@ class GBrowseTrack ( Tabular ):
#TODO: fix me...
return open(dataset.file_name)
def sniff( self, filename ):
"""
Determines whether the file is in gbrowsetrack format.
GBrowseTrack files are built within Galaxy.
TODO: Not yet sure what this file will look like. Fix this sniffer and add some unit tests here as soon as we know.
"""
return ''
if __name__ == '__main__':
import doctest, sys
doctest.testmod(sys.modules[__name__])
+61 -18
View File
@@ -8,11 +8,12 @@ import galaxy.util
from galaxy.util.odict import odict
class Registry( object ):
def __init__( self, datatypes = [] ):
def __init__( self, datatypes=[], sniff_order=[] ):
self.log = logging.getLogger(__name__)
self.datatypes_by_extension = {}
self.mimetypes_by_extension = {}
self.datatype_converters = odict()
self.sniff_order = []
for ext, kind in datatypes:
try:
mime_type = None
@@ -38,8 +39,7 @@ class Registry( object ):
self.datatypes_by_extension = {
'data' : data.Data(),
'bed' : interval.Bed(),
'txt' : data.Text(),
'text' : data.Text(),
'txt' : data.Text(),
'interval' : interval.Interval(),
'tabular' : tabular.Tabular(),
'png' : images.Image(),
@@ -54,14 +54,12 @@ class Registry( object ):
'laj' : images.Laj(),
'lav' : sequence.Lav(),
'html' : images.Html(),
'customtrack' : interval.CustomTrack(),
'gbrowsetrack' : interval.GBrowseTrack()
'customtrack' : interval.CustomTrack()
}
self.mimetypes_by_extension = {
'data' : 'application/octet-stream',
'bed' : 'text/plain',
'txt' : 'text/plain',
'text' : 'text/plain',
'txt' : 'text/plain',
'interval' : 'text/plain',
'tabular' : 'text/plain',
'png' : 'image/png',
@@ -76,14 +74,63 @@ class Registry( object ):
'laj' : 'text/plain',
'lav' : 'text/plain',
'html' : 'text/html',
'customtrack' : 'text/plain',
'gbrowsetrack' : 'text/plain'
'customtrack' : 'text/plain'
}
"""
The order in which we attempt to determine data types is critical
because some formats are much more flexibly defined than others.
"""
sniff_order.sort()
for ord, kind in sniff_order:
try:
fields = kind.split(":")
datatype_module = fields[0]
datatype_class = fields[1]
fields = datatype_module.split(".")
module = __import__(fields.pop(0))
for mod in fields: module = getattr(module,mod)
aclass = getattr(module, datatype_class)()
included = False
for atype in self.sniff_order:
if isinstance(atype, aclass.__class__):
included = True
break
if not included:
self.sniff_order.append(aclass)
except Exception, exc:
self.log.warning('error appending datatype: %s to sniff_order, error: %s' %(str(kind), str(exc)))
#default values
if len(self.sniff_order) < 1:
self.sniff_order = [
images.Gmaj(),
images.Laj(),
sequence.Maf(),
sequence.Lav(),
sequence.Fasta(),
interval.Wiggle(),
images.Html(),
sequence.Axt(),
interval.Bed(),
interval.CustomTrack(),
interval.Gff(),
interval.Gff3(),
interval.Interval()
]
def append_to_sniff_order():
"""Just in case any supported data types are not included in the config's sniff_order section."""
for ext in self.datatypes_by_extension:
datatype = self.datatypes_by_extension[ext]
included = False
for atype in self.sniff_order:
if isinstance(atype, datatype.__class__):
included = True
break
if not included:
self.sniff_order.append(datatype)
append_to_sniff_order()
def get_mimetype_by_extension(self, ext ):
"""
Returns a mimetype based on an extension
"""
"""Returns a mimetype based on an extension"""
try:
mimetype = self.mimetypes_by_extension[ext]
except KeyError:
@@ -93,9 +140,7 @@ class Registry( object ):
return mimetype
def get_datatype_by_extension(self, ext ):
"""
Returns a datatype based on an extension
"""
"""Returns a datatype based on an extension"""
try:
builder = self.datatypes_by_extension[ext]
except KeyError:
@@ -114,9 +159,7 @@ class Registry( object ):
return data
def old_change_datatype(self, data, ext):
"""
Creates and returns a new datatype based on an existing data and an extension
"""
"""Creates and returns a new datatype based on an existing data and an extension"""
newdata = factory(ext)(id=data.id)
for key, value in data.__dict__.items():
setattr(newdata, key, value)
+128 -1
View File
@@ -7,6 +7,7 @@ import logging
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes import metadata
from galaxy import util
from sniff import *
log = logging.getLogger(__name__)
@@ -22,6 +23,7 @@ class Alignment( Sequence ):
class Fasta( Sequence ):
"""Class representing a FASTA sequence"""
file_ext = "fasta"
def set_peek( self, dataset ):
Sequence.set_peek( self, dataset )
@@ -37,14 +39,42 @@ class Fasta( Sequence ):
else:
dataset.blurb = '%d sequences' % count
def sniff(self, filename):
"""
Determines whether the file is in fasta format
A sequence in FASTA format consists of a single-line description, followed by lines of sequence data.
The first character of the description line is a greater-than (">") symbol in the first column.
All lines should be shorter than 80 charcters
For complete details see http://www.g2l.bio.uni-goettingen.de/blast/fastades.html
>>> fname = get_test_fname( 'sequence.maf' )
>>> Fasta().sniff( fname )
''
>>> fname = get_test_fname( 'sequence.fasta' )
>>> Fasta().sniff( fname )
'fasta'
"""
headers = get_headers( filename, None )
try:
if len(headers) > 1 and headers[0][0] and headers[0][0][0] == ">":
return self.file_ext
else:
return ''
except:
return ''
try:
import pkg_resources; pkg_resources.require( "bx-python" )
import bx.align.maf
except:
pass
class Maf( Alignment ):
"""Class describing a Maf alignment"""
file_ext = "maf"
def init_meta( self, dataset, copy_from=None ):
Alignment.init_meta( self, dataset, copy_from=copy_from )
@@ -77,9 +107,106 @@ class Maf( Alignment ):
return True
return False
def sniff( self, filename ):
"""
Determines wether the file is in maf format
The .maf format is line-oriented. Each multiple alignment ends with a blank line.
Each sequence in an alignment is on a single line, which can get quite long, but
there is no length limit. Words in a line are delimited by any white space.
Lines starting with # are considered to be comments. Lines starting with ## can
be ignored by most programs, but contain meta-data of one form or another.
The first line of a .maf file begins with ##maf. This word is followed by white-space-separated
variable=value pairs. There should be no white space surrounding the "=".
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format5
>>> fname = get_test_fname( 'sequence.maf' )
>>> Maf().sniff( fname )
'maf'
>>> fname = get_test_fname( 'sequence.fasta' )
>>> Maf().sniff( fname )
''
"""
headers = get_headers( filename, None )
try:
if len(headers) > 1 and headers[0][0] and headers[0][0] == "##maf":
return self.file_ext
else:
return ''
except:
return ''
class Axt( Alignment ):
"""Class describing an axt alignment"""
file_ext = "axt"
def sniff( self, filename ):
"""
Determines whether the file is in axt format
axt alignment files are produced from Blastz, an alignment tool available from Webb Miller's lab
at Penn State University. Each alignment block in an axt file contains three lines: a summary
line and 2 sequence lines. Blocks are separated from one another by blank lines.
The summary line contains chromosomal position and size information about the alignment. It consists of 9 required fields:
For complete details see http://genome.ucsc.edu/goldenPath/help/axt.html
>>> fname = get_test_fname( 'alignment.axt' )
>>> Axt().sniff( fname )
'axt'
>>> fname = get_test_fname( 'alignment.lav' )
>>> Axt().sniff( fname )
''
"""
headers = get_headers( filename, None )
if len(headers) < 4:
return ''
try:
"""Assume the summary line is the first line of the file."""
line = headers[0]
except:
return ''
if len(line) != 9:
return ''
try:
map ( int, [line[0], line[2], line[3], line[5], line[6], line[8]] )
except:
return ''
if line[7] not in data.valid_strand:
return ''
return self.file_ext
class Lav( Alignment ):
"""Class describing a LAV alignment"""
file_ext = "lav"
def sniff( self, filename ):
"""
Determines whether the file is in lav format
LAV is an alignment format developed by Webb Miller's group. It is the primary output format for BLASTZ.
The first line of a .lav file begins with #:lav.
For complete details see http://www.bioperl.org/wiki/LAV_alignment_format
>>> fname = get_test_fname( 'alignment.lav' )
>>> Lav().sniff( fname )
'lav'
>>> fname = get_test_fname( 'alignment.axt' )
>>> Lav().sniff( fname )
''
"""
headers = get_headers( filename, None )
try:
if len(headers) > 1 and headers[0][0] and headers[0][0].startswith('#:lav'):
return self.file_ext
else:
return ''
except:
return ''
+33 -453
View File
@@ -2,12 +2,9 @@
File format detector
"""
import logging, sys, os, csv, tempfile, shutil, re
import registry
log = logging.getLogger(__name__)
valid_strand = ['+', '-', '.']
valid_gff3_strand = ['+', '-', '.', '?']
valid_gff3_phase = ['.', '0', '1', '2']
def get_test_fname(fname):
"""Returns test data filename"""
@@ -115,416 +112,18 @@ def is_column_based(fname, sep='\t', skip=0):
if not headers:
return False
for hdr in headers[skip:]:
if len(hdr) > 1 and hdr[0] and not hdr[0].startswith('#'):
count = len(hdr)
break
if count < 2:
return False
for hdr in headers[skip:]:
if len(hdr) > 1 and hdr[0] and not hdr[0].startswith('#') and len(hdr) != count:
return False
return True
def is_fasta(headers):
"""
Determines wether the file is in fasta format
A sequence in FASTA format consists of a single-line description, followed by lines of sequence data.
The first character of the description line is a greater-than (">") symbol in the first column.
All lines should be shorter than 80 charcters
For complete details see http://www.g2l.bio.uni-goettingen.de/blast/fastades.html
>>> headers = get_headers(__file__, ' ')
>>> is_fasta(headers)
False
>>> fname = get_test_fname('sequence.fasta')
>>> headers = get_headers(fname,' ')
>>> is_fasta(headers)
True
"""
try:
return len(headers) > 1 and headers[0][0] and headers[0][0][0] == ">"
except:
return False
def is_gff(headers):
"""
Determines wether the file is in gff format
GFF lines have nine required fields that must be tab-separated.
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format3
>>> headers = get_headers(__file__,' ')
>>> is_fasta(headers)
False
>>> fname = get_test_fname('test.gff')
>>> headers = get_headers(fname,'\\t')
>>> is_gff(headers)
True
"""
try:
if len(headers) < 2:
return False
for hdr in headers:
if len( hdr ) > 1 and hdr[0] and not hdr[0].startswith( '#' ):
if len(hdr) != 9:
return False
try:
map( int, [hdr[3], hdr[4]] )
except:
return False
if hdr[5] != '.':
try:
score = int(hdr[5])
except:
return False
if (score < 0 or score > 1000):
return False
if hdr[6] not in valid_strand:
return False
return True
except:
return False
def is_gff3(headers):
"""
Determines wether the file is in gff version 3 format
GFF 3 format:
1) adds a mechanism for representing more than one level
of hierarchical grouping of features and subfeatures.
2) separates the ideas of group membership and feature name/id
3) constrains the feature type field to be taken from a controlled
vocabulary.
4) allows a single feature, such as an exon, to belong to more than
one group at a time.
5) provides an explicit convention for pairwise alignments
6) provides an explicit convention for features that occupy disjunct regions
The format consists of 9 columns, separated by tabs (NOT spaces).
Undefined fields are replaced with the "." character, as described in the original GFF spec.
For complete details see http://song.sourceforge.net/gff3.shtml
>>> headers = get_headers(__file__,' ')
>>> is_fasta(headers)
False
>>> fname = get_test_fname('gff_version_3.gff')
>>> headers = get_headers(fname,'\\t')
>>> is_gff3(headers)
True
"""
try:
if len(headers) < 2:
return False
for hdr in headers:
if len(hdr) > 1 and hdr[0] and not hdr[0].startswith( '#' ):
if len(hdr) != 9:
return False
try:
map( int, [hdr[3]] )
except:
if hdr[3] != '.':
return False
try:
map( int, [hdr[4]] )
except:
if hdr[4] != '.':
return False
if hdr[5] != '.':
try:
score = int(hdr[5])
except:
return False
if (score < 0 or score > 1000):
return False
if hdr[6] not in valid_gff3_strand:
return False
if hdr[7] not in valid_gff3_phase:
return False
return True
except:
return False
def is_maf(headers):
"""
Determines wether the file is in maf format
The .maf format is line-oriented. Each multiple alignment ends with a blank line.
Each sequence in an alignment is on a single line, which can get quite long, but
there is no length limit. Words in a line are delimited by any white space.
Lines starting with # are considered to be comments. Lines starting with ## can
be ignored by most programs, but contain meta-data of one form or another.
The first line of a .maf file begins with ##maf. This word is followed by white-space-separated
variable=value pairs. There should be no white space surrounding the "=".
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format5
>>> headers = get_headers(__file__,' ')
>>> is_maf(headers)
False
>>> fname = get_test_fname('sequence.maf')
>>> headers = get_headers(fname,' ')
>>> is_maf(headers)
True
"""
try:
return len(headers) > 1 and headers[0][0] and headers[0][0] == "##maf"
except:
return False
def is_lav(headers):
"""
Determines wether the file is in lav format
LAV is an alignment format developed by Webb Miller's group. It is the primary output format for BLASTZ.
The first line of a .lav file begins with #:lav.
For complete details see http://www.bioperl.org/wiki/LAV_alignment_format
>>> headers = get_headers(__file__,' ')
>>> is_lav(headers)
False
>>> fname = get_test_fname('alignment.lav')
>>> headers = get_headers(fname,' ')
>>> is_lav(headers)
True
"""
try:
return len(headers) > 1 and headers[0][0] and headers[0][0].startswith('#:lav')
except:
return False
def is_axt(headers):
"""
Determines wether the file is in axt format
axt alignment files are produced from Blastz, an alignment tool available from Webb Miller's lab
at Penn State University. Each alignment block in an axt file contains three lines: a summary
line and 2 sequence lines. Blocks are separated from one another by blank lines.
The summary line contains chromosomal position and size information about the alignment. It consists of 9 required fields:
For complete details see http://genome.ucsc.edu/goldenPath/help/axt.html
>>> headers = get_headers(__file__,None)
>>> is_axt(headers)
False
>>> fname = get_test_fname('alignment.axt')
>>> headers = get_headers(fname,None)
>>> is_axt(headers)
True
"""
try:
#return (len(headers) >= 4) and (headers[0][7] == '-' or headers[0][7] == '+') and (headers[3] == []) and (len(headers[0])==9 or len(headers[0])==10)
if len(headers) < 4:
return False
"""
Assume the summary line is the first line of the file.
"""
line = headers[0]
if len(line) != 9:
return False
try:
map ( int, [line[0], line[2], line[3], line[5], line[6], line[8]] )
except:
return False
if line[7] not in valid_strand:
return False
return True
except:
return False
def is_wiggle(headers):
"""
Determines wether the file is in wiggle format
The .wig format is line-oriented. Wiggle data is preceeded by a track definition line,
which adds a number of options for controlling the default display of this track.
Following the track definition line is the track data, which can be entered in several
different formats.
The track definition line begins with the word 'track' followed by the track type.
The track type with version is REQUIRED, and it currently must be wiggle_0. For example,
track type=wiggle_0...
For complete details see http://genome.ucsc.edu/goldenPath/help/wiggle.html
>>> headers = get_headers(__file__,' ')
>>> is_wiggle(headers)
False
>>> fname = get_test_fname('wiggle.wig')
>>> headers = get_headers(fname,' ')
>>> is_wiggle(headers)
True
"""
try:
for hdr in headers:
if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'):
return True
return False
except:
return False
def is_bed(headers):
"""
Checks for 'bedness'
BED lines have three required fields and nine additional optional fields.
The number of fields per line must be consistent throughout any single set of data in
an annotation track.
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format1
>>> fname = get_test_fname('test_tab.bed')
>>> headers = get_headers(fname,'\\t')
>>> is_bed(headers)
True
>>> fname = get_test_fname('interval.bed')
>>> headers = get_headers(fname,'\\t')
>>> is_bed(headers)
False
>>> fname = get_test_fname('complete.bed')
>>> headers = get_headers(fname,'\\t')
>>> is_bed(headers)
True
"""
col1_startswith = ['chr', 'chl', 'groupun', 'reftig_', 'scaffold', 'super_', 'vcho']
try:
if not headers:
return False
for hdr in headers:
valid_col1 = False
if len(hdr) < 3 or len(hdr) > 12:
return False
for str in col1_startswith:
if hdr[0].lower().startswith(str):
valid_col1 = True
break
if valid_col1:
try:
map( int, [hdr[1], hdr[2]] )
except:
return False
if len(hdr) > 3:
"""
Since all 9 of these fields are optional, it is difficult to test
for specific column values...
"""
optionals = hdr[3:]
"""
...we can, however, test complete BED definitions fairly easily.
"""
if len(optionals) == 9:
try:
map ( int, [optionals[1], optionals[3], optionals[4], optionals[5], optionals[6]] )
except:
return False
score = int(optionals[1])
if score < 0 or score > 1000:
return False
if optionals[2] not in ['+', '-']:
return False
if int(optionals[5]) != 0:
return False
block_count = int(optionals[6])
"""
Sometime the blosck_sizes and block_starts lists end in extra commas
"""
block_sizes = optionals[7].rstrip(',').split(',')
block_starts = optionals[8].rstrip(',').split(',')
if len(block_sizes) != block_count or len(block_starts) != block_count:
return False
elif len(optionals) > 4 and len(optionals) < 9:
"""
Here it gets a bit trickier, but in this case, we can be somewhat confident
that optionals will include a strand column
"""
is_valid_strand = False
for ele in optionals:
if ele in valid_bed_strand:
is_valid_strand = True
if not is_valid_strand:
return False
else:
return False
return True
except:
return False
def is_interval(headers):
"""
Checks for 'intervalness'
This is the most loosely defined format and is mostly used by galaxy itself. In general, if
the format is_column_based, but not any of the other formats, then it must be interval.
>>> fname = get_test_fname('test_space.bed')
>>> headers = get_headers(fname,'\\t')
>>> is_interval(headers)
False
>>> fname = get_test_fname('interval.interval')
>>> headers = get_headers(fname,'\\t')
>>> is_interval(headers)
True
"""
try:
#return is_bed(headers, skip=1) and headers[0][0][0] == '#'
"""
If we got here, we already know the file is_column_based and is not bed,
so we'll just look for some valid data.
"""
for hdr in headers:
if not (hdr[0] == '' or hdr[0].startswith( '#' )):
if len(hdr) < 3:
return False
try:
map( int, [hdr[1], hdr[2]] )
except:
return False
return True
except:
return False
def is_html(headers):
"""
Determines wether the file is in html format
>>> headers = get_headers(__file__,' ')
>>> is_html(headers)
False
>>> fname = get_test_fname('file.html')
>>> headers = get_headers(fname,' ')
>>> is_html(headers)
True
"""
try:
for idx, hdr in enumerate(headers):
if hdr and hdr[0].lower() == '<html>':
return True
if idx > 29:
"""
This is a weakness since it assumes < 29 blank lines, comments, etc.
"""
break
return False
except:
return False
def guess_ext(fname):
def guess_ext( fname ):
"""
Returns an extension that can be used in the datatype factory to
generate a data for the 'fname' file
@@ -552,65 +151,46 @@ def guess_ext(fname):
'gff'
>>> fname = get_test_fname('gff_version_3.gff')
>>> guess_ext(fname)
'gff'
'gff3'
>>> fname = get_test_fname('temp.txt')
>>> file(fname, 'wt').write("a 2\\nc 1")
>>> file(fname, 'wt').write("a 2\\nc 1\\nd 0")
>>> guess_ext(fname)
'tabular'
>>> fname = get_test_fname('temp.txt')
>>> file(fname, 'wt').write("a 1 2 x\\nb 3 4 y")
>>> file(fname, 'wt').write("a 1 2 x\\nb 3 4 y\\nc 5 6 z")
>>> guess_ext(fname)
'interval'
"""
try:
'tabular'
"""
datatypes_registry = registry.Registry()
for datatype in datatypes_registry.sniff_order:
"""
The order in which we attempt to determine data format is pretty important
because some formats are much more flexibly defined than others. Interval format
is the most loosely defined, so it should be that last format we check.
Some classes may not have a sniff function, which is ok. In fact, the
Tabular and Text classes are 2 examples of classes that should never have
a sniff function. Since these classes are default classes, they contain
few rules to filter out data of other formats, so they should be called
from this function after all other datatypes in sniff_order have not been
successfully discovered.
"""
#guess if data is binary
for line in file(fname):
for char in line:
if ord(char) > 128:
return "data"
headers = get_headers(fname, None)
if is_maf(headers):
return 'maf'
elif is_lav(headers):
return 'lav'
elif is_fasta(headers):
return 'fasta'
elif is_wiggle(headers):
return 'wig'
elif is_html(headers):
return 'html'
elif is_axt(headers):
return 'axt'
try:
format = datatype.sniff( fname )
except:
format = ''
if format:
return format
if is_column_based(fname, ' ', 1):
sep2tabs(fname)
if is_column_based(fname, '\t', 0):
headers = get_headers(fname, '\t')
if is_bed(headers):
return 'bed'
if is_column_based(fname, '\t', 1):
headers = get_headers(fname, '\t')
if is_gff(headers):
return 'gff'
if is_gff3(headers):
return 'gff3'
elif is_interval(headers):
return 'interval'
"""Default binary file extension"""
for line in file( fname ):
for char in line:
if ord(char) > 128:
return "data"
else:
return 'tabular'
except:
pass
return 'text'
break
break
if is_column_based( fname, ' ', 1 ):
sep2tabs(fname)
if is_column_based( fname, '\t', 1):
return "tabular"
return 'txt'
if __name__ == '__main__':
import doctest, sys
+30 -101
View File
@@ -6,7 +6,7 @@ import pkg_resources
pkg_resources.require( "bx-python" )
import logging
import data, sniff
import data
from galaxy import util
from cgi import escape
from galaxy.datatypes import metadata
@@ -15,7 +15,6 @@ from galaxy.datatypes.metadata import MetadataAttributes
log = logging.getLogger(__name__)
class Tabular( data.Text ):
"""Tab delimited data"""
@@ -40,118 +39,47 @@ class Tabular( data.Text ):
"""
if dataset.has_data():
column_types = []
format = sniff.guess_ext( dataset.file_name )
if dataset.extension != format:
"""
We need to rely on the ability of our sniffer to properly detect datatypes here since
there are many ways that the datatype could be improperly set.
TODO: we may want to automatically convert the datset to the proper datatype here,
but for now we'll leave it and just rely on the value in format.
"""
pass
"""
The 'proceed' value allows us to skip different numbers of lines based on the data
format (type). We need this because different formats can include lines of information
that are not properly commented (properly commented lines start with a '#' character).
"""
proceed = False
col1_startswith = ['chr', 'chl', 'groupun', 'reftig_', 'scaffold', 'super_', 'vcho']
for i, line in enumerate( file ( dataset.file_name ) ):
line = line.rstrip('\r\n')
valid = True
if line and not line.startswith( '#' ):
elems = line.split( '\t' )
elems_len = len(elems)
if elems_len > 0:
if format == 'bed':
for str in col1_startswith:
if elems[0].lower().startswith(str):
proceed = True
break
elif format == 'interval':
if elems_len > 2:
"""Set the columns metadata attribute"""
if elems_len != dataset.metadata.columns:
dataset.metadata.columns = elems_len
"""Set the column_types metadata attribute"""
for col in range(0, elems_len):
col_type = None
val = elems[col]
if not col_type:
"""See if val is an int"""
try:
map( int, [elems[1], elems[2]] )
proceed = True
except:
pass # proceed is False
elif format == 'gff':
if elems_len == 9:
try:
map( int, [hdr[3], hdr[4]] )
proceed = True
except:
int( val )
col_type = 'int'
except:
pass
elif format == 'gff3':
valid_gff3_strand = ['+', '-', '.', '?']
valid_start = False
valid_end = False
if elems_len == 9:
if not col_type:
"""See if val is a float"""
try:
start = int(hdr[3])
valid_start = True
float( val )
col_type = 'float'
except:
if hdr[3] == '.':
valid_start = True
try:
end = int(hdr[4])
valid_end = True
except:
if hdr[4] == '.':
valid_end = True
srand = hdr[6]
if valid_start and valid_end and start < end and strand in valid_gff3_strand:
proceed = True
elif format=='wig':
try:
int( elems[0] )
proceed = True
except:
for str in col1_startswith:
if elems[0].lower().startswith(str):
proceed = True
break
elif ( format =='tabular' or format == "customtrack" or format == 'gbrowsetrack' ) and i > 1:
proceed = True
if proceed:
"""Set the columns metadata attribute"""
if elems_len != dataset.metadata.columns:
dataset.metadata.columns = elems_len
"""Set the column_types metadata attribute"""
for col in range(0, elems_len):
col_type = None
val = elems[col]
if not col_type:
"""See if val is an int"""
try:
int( val )
col_type = 'int'
except:
pass
if not col_type:
"""See if val is a float"""
try:
float( val )
if val and val.strip().lower() == 'na':
col_type = 'float'
except:
if val and val.strip().lower() == 'na':
col_type = 'float'
if not col_type:
"""See if val is a list"""
val_elems = val.split(',')
if len( val_elems ) > 1:
col_type = 'list'
if not col_type:
"""All parameters are strings, so this will be the default"""
col_type = 'str'
if not col_type:
"""See if val is a list"""
val_elems = val.split(',')
if len( val_elems ) > 1:
col_type = 'list'
if not col_type:
"""All parameters are strings, so this will be the default"""
col_type = 'str'
column_types.append(col_type)
column_types.append(col_type)
if len(column_types) > 0: break
if i > 100: break # Hopefully we never get here...
@@ -204,4 +132,5 @@ class Tabular( data.Text ):
def before_edit( self, dataset ):
data.Text.before_edit( self, dataset )
if self.missing_meta( dataset ):
self.set_meta( dataset )
self.set_meta( dataset )
+4 -4
View File
@@ -1,7 +1,7 @@
# gff-version 2
## Date: Thu Dec 8 19:46:27 2005
## gff.pl $Rev: 601 $
## Input file: /depot/data2/galaxy/encode-data/datasets/msa.AR.20051208.bed
##gff-version 2
##Date: Thu Dec 8 19:46:27 2005
##gff.pl $Rev: 601 $
##Input file: /depot/data2/galaxy/encode-data/datasets/msa.AR.20051208.bed
chr7 bed2gff AR 26731313 26731437 . + . score
chr7 bed2gff AR 26731491 26731536 . + . score
+18 -2
View File
@@ -105,7 +105,6 @@ static_style_dir = %(here)s/static/june_2007_style/blue
data = galaxy.datatypes.data:Data,application/octet-stream
bed = galaxy.datatypes.interval:Bed
txt = galaxy.datatypes.data:Text
text = galaxy.datatypes.data:Text
interval = galaxy.datatypes.interval:Interval
tabular = galaxy.datatypes.tabular:Tabular
png = galaxy.datatypes.images:Image,image/png
@@ -123,7 +122,6 @@ laj = galaxy.datatypes.images:Laj
lav = galaxy.datatypes.sequence:Lav
html = galaxy.datatypes.images:Html,text/html
customtrack = galaxy.datatypes.interval:CustomTrack
gbrowsetrack = galaxy.datatypes.interval:GBrowseTrack
#EMBOSS TOOLS
match = galaxy.datatypes.data:Text
genbank = galaxy.datatypes.data:Text
@@ -171,3 +169,21 @@ swiss = galaxy.datatypes.data:Text
phylipnon = galaxy.datatypes.data:Text
nexusnon = galaxy.datatypes.data:Text
nametable = galaxy.datatypes.data:Text
# ---- Data Type Sniff Order --------------------------------------------------
[galaxy:sniff_order]
05 = galaxy.datatypes.images:Gmaj
10 = galaxy.datatypes.images:Laj
15 = galaxy.datatypes.sequence:Maf
20 = galaxy.datatypes.sequence:Lav
25 = galaxy.datatypes.sequence:Fasta
30 = galaxy.datatypes.interval:Wiggle
35 = galaxy.datatypes.images:Html
40 = galaxy.datatypes.sequence:Axt
45 = galaxy.datatypes.interval:Bed
50 = galaxy.datatypes.interval:CustomTrack
55 = galaxy.datatypes.interval:Gff
60 = galaxy.datatypes.interval:Gff3
65 = galaxy.datatypes.interval:Interval