mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Streamlined interface for supporting new data types - requires db change and note config change.
db change: update dataset set extension = 'txt' where extension = 'text'; Adding support for a new data type is now as simple as adding the new type to the config file and optionally adding the type into the new sniff_order section in the config file, and adding the new class for the data type to the appropriate datatypes code. The config file now has a new section [galaxy:sniff_order] which lists the order in which datatype class sniff functions will be called (each datatype class can now optionally include its own sniff function, although not a requirement). Support for the 'text' file extension has been eliminated as we will standardize on the 'txt' file extension for the Text data type.
This commit is contained in:
+1
-1
@@ -14,7 +14,7 @@ class UniverseApplication( object ):
|
||||
self.config.check()
|
||||
config.configure_logging( self.config )
|
||||
#Set up datatypes registry
|
||||
self.datatypes_registry = galaxy.datatypes.registry.Registry(datatypes = self.config.datatypes)
|
||||
self.datatypes_registry = galaxy.datatypes.registry.Registry(datatypes=self.config.datatypes, sniff_order=self.config.sniff_order)
|
||||
galaxy.model.set_datatypes_registry(self.datatypes_registry)
|
||||
# Connect up the object model
|
||||
if self.config.database_connection:
|
||||
|
||||
@@ -60,6 +60,11 @@ class Configuration( object ):
|
||||
self.datatypes = global_conf_parser.items("galaxy:datatypes")
|
||||
except ConfigParser.NoSectionError:
|
||||
self.datatypes = []
|
||||
#Store sniff order config
|
||||
try:
|
||||
self.sniff_order = global_conf_parser.items("galaxy:sniff_order")
|
||||
except ConfigParser.NoSectionError:
|
||||
self.sniff_order = []
|
||||
self.datatype_converters_config = kwargs.get( 'datatype_converters_config_file', "datatype_converters_conf.xml" )
|
||||
self.datatype_converters_path = kwargs.get( 'datatype_converters_path', os.path.join(self.root,"lib/galaxy/datatypes/converters") )
|
||||
def get( self, key, default ):
|
||||
|
||||
@@ -2,10 +2,14 @@ import logging, os, sys, time, sets, tempfile
|
||||
from galaxy import util
|
||||
from cgi import escape
|
||||
from galaxy.datatypes.metadata import *
|
||||
log = logging.getLogger(__name__)
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes import metadata
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# Valid strand column values
|
||||
valid_strand = ['+', '-', '.']
|
||||
|
||||
# Constants for data states
|
||||
DATA_NEW, DATA_OK, DATA_FAKE = 'new', 'ok', 'fake'
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@ Image classes
|
||||
|
||||
import data
|
||||
import logging
|
||||
from galaxy.datatypes.sniff import *
|
||||
from urllib import urlencode
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -17,6 +18,8 @@ class Image( data.Data ):
|
||||
|
||||
class Gmaj( data.Data ):
|
||||
"""Class describing a GMAJ Applet"""
|
||||
file_ext = "gmaj.zip"
|
||||
|
||||
def set_peek( self, dataset ):
|
||||
dataset.peek = "<p align=\"center\"><applet code=\"edu.psu.bx.gmaj.MajApplet.class\" archive=\"/static/gmaj/gmaj.jar\" width=\"200\" height=\"30\" align=\"middle\"> <param name=bundle value=\"display?id="+str(dataset.id)+"&tofile=yes&toext=.zip\"> <param name=buttonlabel value=\"Launch GMAJ\"><param name=nobutton value=\"false\"><param name=urlpause value=\"100\"><param name=debug value=\"false\"><i>Your browser is not responding to the <applet> tag.</i></applet></p>"
|
||||
dataset.blurb = 'GMAJ Multiple Alignment Viewer'
|
||||
@@ -29,27 +32,57 @@ class Gmaj( data.Data ):
|
||||
def get_mime(self):
|
||||
"""Returns the mime type of the datatype"""
|
||||
return 'application/zip'
|
||||
def sniff( self, filename ):
|
||||
#TODO: fix me
|
||||
return ''
|
||||
|
||||
class Html( data.Text ):
|
||||
"""Class describing an html file"""
|
||||
file_ext = "html"
|
||||
|
||||
def set_peek( self, dataset ):
|
||||
dataset.peek = "HTML file (%s)" % ( data.nice_size( dataset.get_size() ) )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
|
||||
def get_mime(self):
|
||||
"""Returns the mime type of the datatype"""
|
||||
return 'text/html'
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Determines wether the file is in html format
|
||||
|
||||
>>> fname = get_test_fname( 'complete.bed' )
|
||||
>>> Html().sniff( fname )
|
||||
''
|
||||
>>> fname = get_test_fname( 'file.html' )
|
||||
>>> Html().sniff( fname )
|
||||
'html'
|
||||
"""
|
||||
headers = get_headers( filename, None )
|
||||
|
||||
try:
|
||||
for i, hdr in enumerate(headers):
|
||||
if hdr and hdr[0].lower().find( '<html>' ) >=0:
|
||||
return self.file_ext
|
||||
return ''
|
||||
except:
|
||||
return ''
|
||||
|
||||
class Laj( data.Text ):
|
||||
"""Class describing a LAJ Applet"""
|
||||
file_ext = "laj"
|
||||
|
||||
def set_peek( self, dataset ):
|
||||
export_url = "/history_add_to?"+urlencode({'history_id':dataset.history_id,'ext':'lav','name':'LAJ Output','info':'Added by LAJ','dbkey':dataset.dbkey})
|
||||
dataset.peek = "<p align=\"center\"><applet code=\"edu.psu.cse.bio.laj.LajApplet.class\" archive=\"static/laj/laj.jar\" width=\"200\" height=\"30\"><param name=buttonlabel value=\"Launch LAJ\"><param name=title value=\"LAJ in Galaxy\"><param name=posturl value=\""+export_url+"\"><param name=alignfile1 value=\"display?id="+str(dataset.id)+"\"><param name=noseq value=\"true\"></applet></p>"
|
||||
dataset.blurb = 'LAJ Multiple Alignment Viewer'
|
||||
|
||||
def display_peek(self, dataset):
|
||||
try:
|
||||
return dataset.peek
|
||||
except:
|
||||
return "peek unavailable"
|
||||
|
||||
class Html( data.Text ):
|
||||
"""Class describing an html file"""
|
||||
def set_peek( self, dataset ):
|
||||
dataset.peek = "HTML file (%s)" % ( data.nice_size( dataset.get_size() ) )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
|
||||
def get_mime(self):
|
||||
"""Returns the mime type of the datatype"""
|
||||
return 'text/html'
|
||||
def sniff( self, filename ):
|
||||
#TODO: fix me...
|
||||
return ''
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@ pkg_resources.require( "bx-python" )
|
||||
import logging, os, sys, time, sets, tempfile, shutil
|
||||
import data
|
||||
from galaxy import util
|
||||
from galaxy.datatypes.sniff import *
|
||||
from cgi import escape
|
||||
import urllib
|
||||
from bx.intervals.io import *
|
||||
@@ -36,6 +37,7 @@ for key, value in alias_spec.items():
|
||||
|
||||
class Interval( Tabular ):
|
||||
"""Tab delimited data containing interval information"""
|
||||
file_ext = "interval"
|
||||
|
||||
"""Add metadata elements"""
|
||||
MetadataElement( name="chromCol", desc="Chrom column", param=metadata.ColumnParameter )
|
||||
@@ -44,7 +46,6 @@ class Interval( Tabular ):
|
||||
MetadataElement( name="strandCol", desc="Strand column (click box & select)", param=metadata.ColumnParameter, optional=True, no_value=0 )
|
||||
MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True, visible=False )
|
||||
|
||||
|
||||
def __init__(self, **kwd):
|
||||
"""Initialize interval datatype, by adding UCSC display apps"""
|
||||
Tabular.__init__(self, **kwd)
|
||||
@@ -179,8 +180,41 @@ class Interval( Tabular ):
|
||||
"""Return options for removing errors along with a description"""
|
||||
return [("lines","Remove erroneous lines")]
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Checks for 'intervalness'
|
||||
|
||||
This format is mostly used by galaxy itself. Valid interval files should include
|
||||
a valid header comment, but this seems to be loosely regulated.
|
||||
|
||||
>>> fname = get_test_fname( 'test_space.bed' )
|
||||
>>> Interval().sniff( fname )
|
||||
''
|
||||
>>> fname = get_test_fname( 'interval.interval' )
|
||||
>>> Interval().sniff( fname )
|
||||
'interval'
|
||||
"""
|
||||
headers = get_headers( filename, '\t' )
|
||||
try:
|
||||
"""
|
||||
If we got here, we already know the file is_column_based and is not bed,
|
||||
so we'll just look for some valid data.
|
||||
"""
|
||||
for hdr in headers:
|
||||
if not (hdr[0] == '' or hdr[0].startswith( '#' )):
|
||||
if len(hdr) < 3:
|
||||
return ''
|
||||
try:
|
||||
map( int, [hdr[1], hdr[2]] )
|
||||
except:
|
||||
return ''
|
||||
return self.file_ext
|
||||
except:
|
||||
return ''
|
||||
|
||||
class Bed( Interval ):
|
||||
"""Tab delimited data in BED format"""
|
||||
file_ext = "bed"
|
||||
|
||||
"""Add metadata elements"""
|
||||
MetadataElement( name="chromCol", default=1, desc="Chrom column", param=metadata.ColumnParameter )
|
||||
@@ -264,8 +298,93 @@ class Bed( Interval ):
|
||||
try: return open(dataset.file_name)
|
||||
except: return "This item contains no content"
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Checks for 'bedness'
|
||||
|
||||
BED lines have three required fields and nine additional optional fields.
|
||||
The number of fields per line must be consistent throughout any single set of data in
|
||||
an annotation track.
|
||||
|
||||
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format1
|
||||
|
||||
>>> fname = get_test_fname( 'test_tab.bed' )
|
||||
>>> Bed().sniff( fname )
|
||||
'bed'
|
||||
>>> fname = get_test_fname( 'interval.bed' )
|
||||
>>> Bed().sniff( fname )
|
||||
''
|
||||
>>> fname = get_test_fname( 'complete.bed' )
|
||||
>>> Bed().sniff( fname )
|
||||
'bed'
|
||||
"""
|
||||
col1_startswith = ['chr', 'chl', 'groupun', 'reftig_', 'scaffold', 'super_', 'vcho']
|
||||
headers = get_headers( filename, '\t' )
|
||||
try:
|
||||
if not headers:
|
||||
return ''
|
||||
for hdr in headers:
|
||||
valid_col1 = False
|
||||
if len(hdr) < 3 or len(hdr) > 12:
|
||||
return ''
|
||||
for str in col1_startswith:
|
||||
if hdr[0].lower().startswith(str):
|
||||
valid_col1 = True
|
||||
break
|
||||
if valid_col1:
|
||||
try:
|
||||
map( int, [hdr[1], hdr[2]] )
|
||||
except:
|
||||
return ''
|
||||
if len(hdr) > 3:
|
||||
"""
|
||||
Since all 9 of these fields are optional, it is difficult to test
|
||||
for specific column values...
|
||||
"""
|
||||
optionals = hdr[3:]
|
||||
"""
|
||||
...we can, however, test complete BED definitions fairly easily.
|
||||
"""
|
||||
if len(optionals) == 9:
|
||||
try:
|
||||
map ( int, [optionals[1], optionals[3], optionals[4], optionals[5], optionals[6]] )
|
||||
except:
|
||||
return ''
|
||||
score = int(optionals[1])
|
||||
if score < 0 or score > 1000:
|
||||
return ''
|
||||
if optionals[2] not in ['+', '-']:
|
||||
return ''
|
||||
if int(optionals[5]) != 0:
|
||||
return ''
|
||||
block_count = int(optionals[6])
|
||||
"""
|
||||
Sometimes the blosck_sizes and block_starts lists end in extra commas
|
||||
"""
|
||||
block_sizes = optionals[7].rstrip(',').split(',')
|
||||
block_starts = optionals[8].rstrip(',').split(',')
|
||||
if len(block_sizes) != block_count or len(block_starts) != block_count:
|
||||
return ''
|
||||
elif len(optionals) > 4 and len(optionals) < 9:
|
||||
"""
|
||||
Here it gets a bit trickier, but in this case, we can be somewhat confident
|
||||
that optionals will include a strand column
|
||||
"""
|
||||
is_valid_strand = False
|
||||
for ele in optionals:
|
||||
if ele in data.valid_strand:
|
||||
is_valid_strand = True
|
||||
if not is_valid_strand:
|
||||
return ''
|
||||
else:
|
||||
return ''
|
||||
return self.file_ext
|
||||
except:
|
||||
return ''
|
||||
|
||||
class Gff( Tabular ):
|
||||
"""Tab delimited data in Gff format"""
|
||||
file_ext = "gff"
|
||||
|
||||
"""Add metadata elements"""
|
||||
MetadataElement( name="columns", default=9, desc="Number of columns", readonly=True )
|
||||
@@ -330,8 +449,51 @@ class Gff( Tabular ):
|
||||
ret_val.append( (site_name, link) )
|
||||
return ret_val
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Determines whether the file is in gff format
|
||||
|
||||
GFF lines have nine required fields that must be tab-separated.
|
||||
|
||||
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format3
|
||||
|
||||
>>> fname = get_test_fname( 'gff_version_3.gff' )
|
||||
>>> Gff().sniff( fname )
|
||||
''
|
||||
>>> fname = get_test_fname( 'test.gff' )
|
||||
>>> Gff().sniff( fname )
|
||||
'gff'
|
||||
"""
|
||||
headers = get_headers( filename, '\t' )
|
||||
try:
|
||||
if len(headers) < 2:
|
||||
return ''
|
||||
for hdr in headers:
|
||||
if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '2' ) < 0:
|
||||
return ''
|
||||
if hdr and hdr[0] and not hdr[0].startswith( '#' ):
|
||||
if len(hdr) != 9:
|
||||
return ''
|
||||
try:
|
||||
map( int, [hdr[3], hdr[4]] )
|
||||
except:
|
||||
return ''
|
||||
if hdr[5] != '.':
|
||||
try:
|
||||
score = int(hdr[5])
|
||||
except:
|
||||
return ''
|
||||
if (score < 0 or score > 1000):
|
||||
return ''
|
||||
if hdr[6] not in data.valid_strand:
|
||||
return ''
|
||||
return self.file_ext
|
||||
except:
|
||||
return ''
|
||||
|
||||
class Gff3( Gff ):
|
||||
"""Tab delimited data in Gff3 format"""
|
||||
file_ext = "gff3"
|
||||
|
||||
"""Add metadata elements"""
|
||||
MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], desc="Column types", readonly=True )
|
||||
@@ -340,18 +502,116 @@ class Gff3( Gff ):
|
||||
"""Initialize datatype, by adding GBrowse display app"""
|
||||
Gff.__init__(self, **kwd)
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Determines whether the file is in gff version 3 format
|
||||
|
||||
GFF 3 format:
|
||||
|
||||
1) adds a mechanism for representing more than one level
|
||||
of hierarchical grouping of features and subfeatures.
|
||||
2) separates the ideas of group membership and feature name/id
|
||||
3) constrains the feature type field to be taken from a controlled
|
||||
vocabulary.
|
||||
4) allows a single feature, such as an exon, to belong to more than
|
||||
one group at a time.
|
||||
5) provides an explicit convention for pairwise alignments
|
||||
6) provides an explicit convention for features that occupy disjunct regions
|
||||
|
||||
The format consists of 9 columns, separated by tabs (NOT spaces).
|
||||
|
||||
Undefined fields are replaced with the "." character, as described in the original GFF spec.
|
||||
|
||||
For complete details see http://song.sourceforge.net/gff3.shtml
|
||||
|
||||
>>> fname = get_test_fname( 'test.gff' )
|
||||
>>> Gff3().sniff( fname )
|
||||
''
|
||||
>>> fname = get_test_fname('gff_version_3.gff')
|
||||
>>> Gff3().sniff( fname )
|
||||
'gff3'
|
||||
"""
|
||||
valid_gff3_strand = ['+', '-', '.', '?']
|
||||
valid_gff3_phase = ['.', '0', '1', '2']
|
||||
headers = get_headers( filename, '\t' )
|
||||
try:
|
||||
if len(headers) < 2:
|
||||
return ''
|
||||
for hdr in headers:
|
||||
if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '3' ) < 0:
|
||||
return ''
|
||||
if hdr and hdr[0] and not hdr[0].startswith( '#' ):
|
||||
if len(hdr) != 9:
|
||||
return ''
|
||||
try:
|
||||
map( int, [hdr[3]] )
|
||||
except:
|
||||
if hdr[3] != '.':
|
||||
return ''
|
||||
try:
|
||||
map( int, [hdr[4]] )
|
||||
except:
|
||||
if hdr[4] != '.':
|
||||
return ''
|
||||
if hdr[5] != '.':
|
||||
try:
|
||||
score = int(hdr[5])
|
||||
except:
|
||||
return ''
|
||||
if (score < 0 or score > 1000):
|
||||
return ''
|
||||
if hdr[6] not in valid_gff3_strand:
|
||||
return ''
|
||||
if hdr[7] not in valid_gff3_phase:
|
||||
return ''
|
||||
return self.file_ext
|
||||
except:
|
||||
return ''
|
||||
|
||||
class Wiggle( Tabular ):
|
||||
"""Tab delimited data in wiggle format"""
|
||||
file_ext = "wig"
|
||||
|
||||
MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True )
|
||||
|
||||
def make_html_table(self, data):
|
||||
return Tabular.make_html_table(self, data, skipchar='#')
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Determines wether the file is in wiggle format
|
||||
|
||||
#Extend Tabular type, since interval tools will fail on track def line (we should fix this)
|
||||
#This is a skeleton class for now, allows viewing at ucsc and formatted peeking.
|
||||
The .wig format is line-oriented. Wiggle data is preceeded by a track definition line,
|
||||
which adds a number of options for controlling the default display of this track.
|
||||
Following the track definition line is the track data, which can be entered in several
|
||||
different formats.
|
||||
|
||||
The track definition line begins with the word 'track' followed by the track type.
|
||||
The track type with version is REQUIRED, and it currently must be wiggle_0. For example,
|
||||
track type=wiggle_0...
|
||||
|
||||
For complete details see http://genome.ucsc.edu/goldenPath/help/wiggle.html
|
||||
|
||||
>>> fname = get_test_fname( 'interval.bed' )
|
||||
>>> Wiggle().sniff( fname )
|
||||
''
|
||||
>>> fname = get_test_fname( 'wiggle.wig' )
|
||||
>>> Wiggle().sniff( fname )
|
||||
'wig'
|
||||
"""
|
||||
headers = get_headers( filename, None )
|
||||
try:
|
||||
for hdr in headers:
|
||||
if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'):
|
||||
return self.file_ext
|
||||
return ''
|
||||
except:
|
||||
return ''
|
||||
|
||||
class CustomTrack ( Tabular ):
|
||||
"""UCSC CustomTrack"""
|
||||
|
||||
file_ext = "customtrack"
|
||||
|
||||
def __init__(self, **kwd):
|
||||
"""Initialize interval datatype, by adding UCSC display app"""
|
||||
Tabular.__init__(self, **kwd)
|
||||
@@ -394,9 +654,49 @@ class CustomTrack ( Tabular ):
|
||||
ret_val.append( (site_name, link) )
|
||||
return ret_val
|
||||
|
||||
#Extend Tabular type, since interval tools will fail on track def line (we should fix this)
|
||||
#This is a skeleton class for now, allows viewing at GBrowse and formatted peeking.
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Determines whether the file is in customtrack format.
|
||||
|
||||
CustomTrack files are built within Galaxy and are basically bed or interval files with the first line looking
|
||||
something like this.
|
||||
|
||||
track name="User Track" description="User Supplied Track (from Galaxy)" color=0,0,0 visibility=1
|
||||
|
||||
>>> fname = get_test_fname( 'complete.bed' )
|
||||
>>> CustomTrack().sniff( fname )
|
||||
''
|
||||
>>> fname = get_test_fname( 'ucsc.customtrack' )
|
||||
>>> CustomTrack().sniff( fname )
|
||||
'customtrack'
|
||||
"""
|
||||
headers = get_headers( filename, None )
|
||||
first_line = True
|
||||
for hdr in headers:
|
||||
if first_line:
|
||||
try:
|
||||
if hdr[0].startswith('track'):
|
||||
first_line = False
|
||||
else:
|
||||
return ''
|
||||
except:
|
||||
return ''
|
||||
else:
|
||||
try:
|
||||
if not (hdr[0] == '' or hdr[0].startswith( '#' )):
|
||||
if len(hdr) < 3:
|
||||
return ''
|
||||
try:
|
||||
map( int, [hdr[1], hdr[2]] )
|
||||
except:
|
||||
return ''
|
||||
except:
|
||||
return ''
|
||||
return self.file_ext
|
||||
|
||||
class GBrowseTrack ( Tabular ):
|
||||
"""GMOD GBrowseTrack"""
|
||||
file_ext = "gbrowsetrack"
|
||||
|
||||
def __init__(self, **kwd):
|
||||
"""Initialize datatype, by adding GBrowse display app"""
|
||||
@@ -431,6 +731,15 @@ class GBrowseTrack ( Tabular ):
|
||||
#TODO: fix me...
|
||||
return open(dataset.file_name)
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Determines whether the file is in gbrowsetrack format.
|
||||
|
||||
GBrowseTrack files are built within Galaxy.
|
||||
TODO: Not yet sure what this file will look like. Fix this sniffer and add some unit tests here as soon as we know.
|
||||
"""
|
||||
return ''
|
||||
|
||||
if __name__ == '__main__':
|
||||
import doctest, sys
|
||||
doctest.testmod(sys.modules[__name__])
|
||||
|
||||
@@ -8,11 +8,12 @@ import galaxy.util
|
||||
from galaxy.util.odict import odict
|
||||
|
||||
class Registry( object ):
|
||||
def __init__( self, datatypes = [] ):
|
||||
def __init__( self, datatypes=[], sniff_order=[] ):
|
||||
self.log = logging.getLogger(__name__)
|
||||
self.datatypes_by_extension = {}
|
||||
self.mimetypes_by_extension = {}
|
||||
self.datatype_converters = odict()
|
||||
self.sniff_order = []
|
||||
for ext, kind in datatypes:
|
||||
try:
|
||||
mime_type = None
|
||||
@@ -38,8 +39,7 @@ class Registry( object ):
|
||||
self.datatypes_by_extension = {
|
||||
'data' : data.Data(),
|
||||
'bed' : interval.Bed(),
|
||||
'txt' : data.Text(),
|
||||
'text' : data.Text(),
|
||||
'txt' : data.Text(),
|
||||
'interval' : interval.Interval(),
|
||||
'tabular' : tabular.Tabular(),
|
||||
'png' : images.Image(),
|
||||
@@ -54,14 +54,12 @@ class Registry( object ):
|
||||
'laj' : images.Laj(),
|
||||
'lav' : sequence.Lav(),
|
||||
'html' : images.Html(),
|
||||
'customtrack' : interval.CustomTrack(),
|
||||
'gbrowsetrack' : interval.GBrowseTrack()
|
||||
'customtrack' : interval.CustomTrack()
|
||||
}
|
||||
self.mimetypes_by_extension = {
|
||||
'data' : 'application/octet-stream',
|
||||
'bed' : 'text/plain',
|
||||
'txt' : 'text/plain',
|
||||
'text' : 'text/plain',
|
||||
'txt' : 'text/plain',
|
||||
'interval' : 'text/plain',
|
||||
'tabular' : 'text/plain',
|
||||
'png' : 'image/png',
|
||||
@@ -76,14 +74,63 @@ class Registry( object ):
|
||||
'laj' : 'text/plain',
|
||||
'lav' : 'text/plain',
|
||||
'html' : 'text/html',
|
||||
'customtrack' : 'text/plain',
|
||||
'gbrowsetrack' : 'text/plain'
|
||||
'customtrack' : 'text/plain'
|
||||
}
|
||||
"""
|
||||
The order in which we attempt to determine data types is critical
|
||||
because some formats are much more flexibly defined than others.
|
||||
"""
|
||||
sniff_order.sort()
|
||||
for ord, kind in sniff_order:
|
||||
try:
|
||||
fields = kind.split(":")
|
||||
datatype_module = fields[0]
|
||||
datatype_class = fields[1]
|
||||
fields = datatype_module.split(".")
|
||||
module = __import__(fields.pop(0))
|
||||
for mod in fields: module = getattr(module,mod)
|
||||
aclass = getattr(module, datatype_class)()
|
||||
included = False
|
||||
for atype in self.sniff_order:
|
||||
if isinstance(atype, aclass.__class__):
|
||||
included = True
|
||||
break
|
||||
if not included:
|
||||
self.sniff_order.append(aclass)
|
||||
except Exception, exc:
|
||||
self.log.warning('error appending datatype: %s to sniff_order, error: %s' %(str(kind), str(exc)))
|
||||
#default values
|
||||
if len(self.sniff_order) < 1:
|
||||
self.sniff_order = [
|
||||
images.Gmaj(),
|
||||
images.Laj(),
|
||||
sequence.Maf(),
|
||||
sequence.Lav(),
|
||||
sequence.Fasta(),
|
||||
interval.Wiggle(),
|
||||
images.Html(),
|
||||
sequence.Axt(),
|
||||
interval.Bed(),
|
||||
interval.CustomTrack(),
|
||||
interval.Gff(),
|
||||
interval.Gff3(),
|
||||
interval.Interval()
|
||||
]
|
||||
def append_to_sniff_order():
|
||||
"""Just in case any supported data types are not included in the config's sniff_order section."""
|
||||
for ext in self.datatypes_by_extension:
|
||||
datatype = self.datatypes_by_extension[ext]
|
||||
included = False
|
||||
for atype in self.sniff_order:
|
||||
if isinstance(atype, datatype.__class__):
|
||||
included = True
|
||||
break
|
||||
if not included:
|
||||
self.sniff_order.append(datatype)
|
||||
append_to_sniff_order()
|
||||
|
||||
def get_mimetype_by_extension(self, ext ):
|
||||
"""
|
||||
Returns a mimetype based on an extension
|
||||
"""
|
||||
"""Returns a mimetype based on an extension"""
|
||||
try:
|
||||
mimetype = self.mimetypes_by_extension[ext]
|
||||
except KeyError:
|
||||
@@ -93,9 +140,7 @@ class Registry( object ):
|
||||
return mimetype
|
||||
|
||||
def get_datatype_by_extension(self, ext ):
|
||||
"""
|
||||
Returns a datatype based on an extension
|
||||
"""
|
||||
"""Returns a datatype based on an extension"""
|
||||
try:
|
||||
builder = self.datatypes_by_extension[ext]
|
||||
except KeyError:
|
||||
@@ -114,9 +159,7 @@ class Registry( object ):
|
||||
return data
|
||||
|
||||
def old_change_datatype(self, data, ext):
|
||||
"""
|
||||
Creates and returns a new datatype based on an existing data and an extension
|
||||
"""
|
||||
"""Creates and returns a new datatype based on an existing data and an extension"""
|
||||
newdata = factory(ext)(id=data.id)
|
||||
for key, value in data.__dict__.items():
|
||||
setattr(newdata, key, value)
|
||||
|
||||
@@ -7,6 +7,7 @@ import logging
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes import metadata
|
||||
from galaxy import util
|
||||
from sniff import *
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -22,6 +23,7 @@ class Alignment( Sequence ):
|
||||
|
||||
class Fasta( Sequence ):
|
||||
"""Class representing a FASTA sequence"""
|
||||
file_ext = "fasta"
|
||||
|
||||
def set_peek( self, dataset ):
|
||||
Sequence.set_peek( self, dataset )
|
||||
@@ -37,14 +39,42 @@ class Fasta( Sequence ):
|
||||
else:
|
||||
dataset.blurb = '%d sequences' % count
|
||||
|
||||
def sniff(self, filename):
|
||||
"""
|
||||
Determines whether the file is in fasta format
|
||||
|
||||
A sequence in FASTA format consists of a single-line description, followed by lines of sequence data.
|
||||
The first character of the description line is a greater-than (">") symbol in the first column.
|
||||
All lines should be shorter than 80 charcters
|
||||
|
||||
For complete details see http://www.g2l.bio.uni-goettingen.de/blast/fastades.html
|
||||
|
||||
>>> fname = get_test_fname( 'sequence.maf' )
|
||||
>>> Fasta().sniff( fname )
|
||||
''
|
||||
>>> fname = get_test_fname( 'sequence.fasta' )
|
||||
>>> Fasta().sniff( fname )
|
||||
'fasta'
|
||||
"""
|
||||
headers = get_headers( filename, None )
|
||||
try:
|
||||
if len(headers) > 1 and headers[0][0] and headers[0][0][0] == ">":
|
||||
return self.file_ext
|
||||
else:
|
||||
return ''
|
||||
except:
|
||||
return ''
|
||||
|
||||
try:
|
||||
import pkg_resources; pkg_resources.require( "bx-python" )
|
||||
import bx.align.maf
|
||||
except:
|
||||
pass
|
||||
|
||||
class Maf( Alignment ):
|
||||
"""Class describing a Maf alignment"""
|
||||
|
||||
file_ext = "maf"
|
||||
|
||||
def init_meta( self, dataset, copy_from=None ):
|
||||
Alignment.init_meta( self, dataset, copy_from=copy_from )
|
||||
|
||||
@@ -77,9 +107,106 @@ class Maf( Alignment ):
|
||||
return True
|
||||
return False
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Determines wether the file is in maf format
|
||||
|
||||
The .maf format is line-oriented. Each multiple alignment ends with a blank line.
|
||||
Each sequence in an alignment is on a single line, which can get quite long, but
|
||||
there is no length limit. Words in a line are delimited by any white space.
|
||||
Lines starting with # are considered to be comments. Lines starting with ## can
|
||||
be ignored by most programs, but contain meta-data of one form or another.
|
||||
|
||||
The first line of a .maf file begins with ##maf. This word is followed by white-space-separated
|
||||
variable=value pairs. There should be no white space surrounding the "=".
|
||||
|
||||
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format5
|
||||
|
||||
>>> fname = get_test_fname( 'sequence.maf' )
|
||||
>>> Maf().sniff( fname )
|
||||
'maf'
|
||||
>>> fname = get_test_fname( 'sequence.fasta' )
|
||||
>>> Maf().sniff( fname )
|
||||
''
|
||||
"""
|
||||
headers = get_headers( filename, None )
|
||||
try:
|
||||
if len(headers) > 1 and headers[0][0] and headers[0][0] == "##maf":
|
||||
return self.file_ext
|
||||
else:
|
||||
return ''
|
||||
except:
|
||||
return ''
|
||||
|
||||
class Axt( Alignment ):
|
||||
"""Class describing an axt alignment"""
|
||||
file_ext = "axt"
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Determines whether the file is in axt format
|
||||
|
||||
axt alignment files are produced from Blastz, an alignment tool available from Webb Miller's lab
|
||||
at Penn State University. Each alignment block in an axt file contains three lines: a summary
|
||||
line and 2 sequence lines. Blocks are separated from one another by blank lines.
|
||||
|
||||
The summary line contains chromosomal position and size information about the alignment. It consists of 9 required fields:
|
||||
|
||||
For complete details see http://genome.ucsc.edu/goldenPath/help/axt.html
|
||||
|
||||
>>> fname = get_test_fname( 'alignment.axt' )
|
||||
>>> Axt().sniff( fname )
|
||||
'axt'
|
||||
>>> fname = get_test_fname( 'alignment.lav' )
|
||||
>>> Axt().sniff( fname )
|
||||
''
|
||||
"""
|
||||
headers = get_headers( filename, None )
|
||||
if len(headers) < 4:
|
||||
return ''
|
||||
try:
|
||||
"""Assume the summary line is the first line of the file."""
|
||||
line = headers[0]
|
||||
except:
|
||||
return ''
|
||||
|
||||
if len(line) != 9:
|
||||
return ''
|
||||
try:
|
||||
map ( int, [line[0], line[2], line[3], line[5], line[6], line[8]] )
|
||||
except:
|
||||
return ''
|
||||
if line[7] not in data.valid_strand:
|
||||
return ''
|
||||
return self.file_ext
|
||||
|
||||
class Lav( Alignment ):
|
||||
"""Class describing a LAV alignment"""
|
||||
file_ext = "lav"
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Determines whether the file is in lav format
|
||||
|
||||
LAV is an alignment format developed by Webb Miller's group. It is the primary output format for BLASTZ.
|
||||
The first line of a .lav file begins with #:lav.
|
||||
|
||||
For complete details see http://www.bioperl.org/wiki/LAV_alignment_format
|
||||
|
||||
>>> fname = get_test_fname( 'alignment.lav' )
|
||||
>>> Lav().sniff( fname )
|
||||
'lav'
|
||||
>>> fname = get_test_fname( 'alignment.axt' )
|
||||
>>> Lav().sniff( fname )
|
||||
''
|
||||
"""
|
||||
headers = get_headers( filename, None )
|
||||
try:
|
||||
if len(headers) > 1 and headers[0][0] and headers[0][0].startswith('#:lav'):
|
||||
return self.file_ext
|
||||
else:
|
||||
return ''
|
||||
except:
|
||||
return ''
|
||||
|
||||
|
||||
|
||||
+33
-453
@@ -2,12 +2,9 @@
|
||||
File format detector
|
||||
"""
|
||||
import logging, sys, os, csv, tempfile, shutil, re
|
||||
import registry
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
valid_strand = ['+', '-', '.']
|
||||
valid_gff3_strand = ['+', '-', '.', '?']
|
||||
valid_gff3_phase = ['.', '0', '1', '2']
|
||||
|
||||
def get_test_fname(fname):
|
||||
"""Returns test data filename"""
|
||||
@@ -115,416 +112,18 @@ def is_column_based(fname, sep='\t', skip=0):
|
||||
|
||||
if not headers:
|
||||
return False
|
||||
|
||||
for hdr in headers[skip:]:
|
||||
if len(hdr) > 1 and hdr[0] and not hdr[0].startswith('#'):
|
||||
count = len(hdr)
|
||||
break
|
||||
|
||||
if count < 2:
|
||||
return False
|
||||
|
||||
for hdr in headers[skip:]:
|
||||
if len(hdr) > 1 and hdr[0] and not hdr[0].startswith('#') and len(hdr) != count:
|
||||
return False
|
||||
return True
|
||||
|
||||
def is_fasta(headers):
|
||||
"""
|
||||
Determines wether the file is in fasta format
|
||||
|
||||
A sequence in FASTA format consists of a single-line description, followed by lines of sequence data.
|
||||
The first character of the description line is a greater-than (">") symbol in the first column.
|
||||
All lines should be shorter than 80 charcters
|
||||
|
||||
For complete details see http://www.g2l.bio.uni-goettingen.de/blast/fastades.html
|
||||
|
||||
>>> headers = get_headers(__file__, ' ')
|
||||
>>> is_fasta(headers)
|
||||
False
|
||||
>>> fname = get_test_fname('sequence.fasta')
|
||||
>>> headers = get_headers(fname,' ')
|
||||
>>> is_fasta(headers)
|
||||
True
|
||||
"""
|
||||
try:
|
||||
return len(headers) > 1 and headers[0][0] and headers[0][0][0] == ">"
|
||||
except:
|
||||
return False
|
||||
|
||||
def is_gff(headers):
|
||||
"""
|
||||
Determines wether the file is in gff format
|
||||
|
||||
GFF lines have nine required fields that must be tab-separated.
|
||||
|
||||
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format3
|
||||
|
||||
>>> headers = get_headers(__file__,' ')
|
||||
>>> is_fasta(headers)
|
||||
False
|
||||
>>> fname = get_test_fname('test.gff')
|
||||
>>> headers = get_headers(fname,'\\t')
|
||||
>>> is_gff(headers)
|
||||
True
|
||||
"""
|
||||
try:
|
||||
if len(headers) < 2:
|
||||
return False
|
||||
for hdr in headers:
|
||||
if len( hdr ) > 1 and hdr[0] and not hdr[0].startswith( '#' ):
|
||||
if len(hdr) != 9:
|
||||
return False
|
||||
try:
|
||||
map( int, [hdr[3], hdr[4]] )
|
||||
except:
|
||||
return False
|
||||
if hdr[5] != '.':
|
||||
try:
|
||||
score = int(hdr[5])
|
||||
except:
|
||||
return False
|
||||
if (score < 0 or score > 1000):
|
||||
return False
|
||||
if hdr[6] not in valid_strand:
|
||||
return False
|
||||
return True
|
||||
except:
|
||||
return False
|
||||
|
||||
def is_gff3(headers):
|
||||
"""
|
||||
Determines wether the file is in gff version 3 format
|
||||
|
||||
GFF 3 format:
|
||||
|
||||
1) adds a mechanism for representing more than one level
|
||||
of hierarchical grouping of features and subfeatures.
|
||||
2) separates the ideas of group membership and feature name/id
|
||||
3) constrains the feature type field to be taken from a controlled
|
||||
vocabulary.
|
||||
4) allows a single feature, such as an exon, to belong to more than
|
||||
one group at a time.
|
||||
5) provides an explicit convention for pairwise alignments
|
||||
6) provides an explicit convention for features that occupy disjunct regions
|
||||
|
||||
The format consists of 9 columns, separated by tabs (NOT spaces).
|
||||
|
||||
Undefined fields are replaced with the "." character, as described in the original GFF spec.
|
||||
|
||||
For complete details see http://song.sourceforge.net/gff3.shtml
|
||||
|
||||
>>> headers = get_headers(__file__,' ')
|
||||
>>> is_fasta(headers)
|
||||
False
|
||||
>>> fname = get_test_fname('gff_version_3.gff')
|
||||
>>> headers = get_headers(fname,'\\t')
|
||||
>>> is_gff3(headers)
|
||||
True
|
||||
"""
|
||||
try:
|
||||
if len(headers) < 2:
|
||||
return False
|
||||
for hdr in headers:
|
||||
if len(hdr) > 1 and hdr[0] and not hdr[0].startswith( '#' ):
|
||||
if len(hdr) != 9:
|
||||
return False
|
||||
try:
|
||||
map( int, [hdr[3]] )
|
||||
except:
|
||||
if hdr[3] != '.':
|
||||
return False
|
||||
try:
|
||||
map( int, [hdr[4]] )
|
||||
except:
|
||||
if hdr[4] != '.':
|
||||
return False
|
||||
if hdr[5] != '.':
|
||||
try:
|
||||
score = int(hdr[5])
|
||||
except:
|
||||
return False
|
||||
if (score < 0 or score > 1000):
|
||||
return False
|
||||
if hdr[6] not in valid_gff3_strand:
|
||||
return False
|
||||
if hdr[7] not in valid_gff3_phase:
|
||||
return False
|
||||
return True
|
||||
except:
|
||||
return False
|
||||
|
||||
def is_maf(headers):
|
||||
"""
|
||||
Determines wether the file is in maf format
|
||||
|
||||
The .maf format is line-oriented. Each multiple alignment ends with a blank line.
|
||||
Each sequence in an alignment is on a single line, which can get quite long, but
|
||||
there is no length limit. Words in a line are delimited by any white space.
|
||||
Lines starting with # are considered to be comments. Lines starting with ## can
|
||||
be ignored by most programs, but contain meta-data of one form or another.
|
||||
|
||||
The first line of a .maf file begins with ##maf. This word is followed by white-space-separated
|
||||
variable=value pairs. There should be no white space surrounding the "=".
|
||||
|
||||
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format5
|
||||
|
||||
>>> headers = get_headers(__file__,' ')
|
||||
>>> is_maf(headers)
|
||||
False
|
||||
>>> fname = get_test_fname('sequence.maf')
|
||||
>>> headers = get_headers(fname,' ')
|
||||
>>> is_maf(headers)
|
||||
True
|
||||
"""
|
||||
try:
|
||||
return len(headers) > 1 and headers[0][0] and headers[0][0] == "##maf"
|
||||
except:
|
||||
return False
|
||||
|
||||
def is_lav(headers):
|
||||
"""
|
||||
Determines wether the file is in lav format
|
||||
|
||||
LAV is an alignment format developed by Webb Miller's group. It is the primary output format for BLASTZ.
|
||||
The first line of a .lav file begins with #:lav.
|
||||
|
||||
For complete details see http://www.bioperl.org/wiki/LAV_alignment_format
|
||||
|
||||
>>> headers = get_headers(__file__,' ')
|
||||
>>> is_lav(headers)
|
||||
False
|
||||
>>> fname = get_test_fname('alignment.lav')
|
||||
>>> headers = get_headers(fname,' ')
|
||||
>>> is_lav(headers)
|
||||
True
|
||||
"""
|
||||
try:
|
||||
return len(headers) > 1 and headers[0][0] and headers[0][0].startswith('#:lav')
|
||||
except:
|
||||
return False
|
||||
|
||||
def is_axt(headers):
|
||||
"""
|
||||
Determines wether the file is in axt format
|
||||
|
||||
axt alignment files are produced from Blastz, an alignment tool available from Webb Miller's lab
|
||||
at Penn State University. Each alignment block in an axt file contains three lines: a summary
|
||||
line and 2 sequence lines. Blocks are separated from one another by blank lines.
|
||||
|
||||
The summary line contains chromosomal position and size information about the alignment. It consists of 9 required fields:
|
||||
|
||||
For complete details see http://genome.ucsc.edu/goldenPath/help/axt.html
|
||||
|
||||
>>> headers = get_headers(__file__,None)
|
||||
>>> is_axt(headers)
|
||||
False
|
||||
>>> fname = get_test_fname('alignment.axt')
|
||||
>>> headers = get_headers(fname,None)
|
||||
>>> is_axt(headers)
|
||||
True
|
||||
"""
|
||||
try:
|
||||
#return (len(headers) >= 4) and (headers[0][7] == '-' or headers[0][7] == '+') and (headers[3] == []) and (len(headers[0])==9 or len(headers[0])==10)
|
||||
if len(headers) < 4:
|
||||
return False
|
||||
"""
|
||||
Assume the summary line is the first line of the file.
|
||||
"""
|
||||
line = headers[0]
|
||||
|
||||
if len(line) != 9:
|
||||
return False
|
||||
try:
|
||||
map ( int, [line[0], line[2], line[3], line[5], line[6], line[8]] )
|
||||
except:
|
||||
return False
|
||||
if line[7] not in valid_strand:
|
||||
return False
|
||||
return True
|
||||
except:
|
||||
return False
|
||||
|
||||
def is_wiggle(headers):
|
||||
"""
|
||||
Determines wether the file is in wiggle format
|
||||
|
||||
The .wig format is line-oriented. Wiggle data is preceeded by a track definition line,
|
||||
which adds a number of options for controlling the default display of this track.
|
||||
Following the track definition line is the track data, which can be entered in several
|
||||
different formats.
|
||||
|
||||
The track definition line begins with the word 'track' followed by the track type.
|
||||
The track type with version is REQUIRED, and it currently must be wiggle_0. For example,
|
||||
track type=wiggle_0...
|
||||
|
||||
For complete details see http://genome.ucsc.edu/goldenPath/help/wiggle.html
|
||||
|
||||
>>> headers = get_headers(__file__,' ')
|
||||
>>> is_wiggle(headers)
|
||||
False
|
||||
>>> fname = get_test_fname('wiggle.wig')
|
||||
>>> headers = get_headers(fname,' ')
|
||||
>>> is_wiggle(headers)
|
||||
True
|
||||
"""
|
||||
try:
|
||||
for hdr in headers:
|
||||
if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'):
|
||||
return True
|
||||
return False
|
||||
except:
|
||||
return False
|
||||
|
||||
def is_bed(headers):
|
||||
"""
|
||||
Checks for 'bedness'
|
||||
|
||||
BED lines have three required fields and nine additional optional fields.
|
||||
The number of fields per line must be consistent throughout any single set of data in
|
||||
an annotation track.
|
||||
|
||||
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format1
|
||||
|
||||
>>> fname = get_test_fname('test_tab.bed')
|
||||
>>> headers = get_headers(fname,'\\t')
|
||||
>>> is_bed(headers)
|
||||
True
|
||||
>>> fname = get_test_fname('interval.bed')
|
||||
>>> headers = get_headers(fname,'\\t')
|
||||
>>> is_bed(headers)
|
||||
False
|
||||
>>> fname = get_test_fname('complete.bed')
|
||||
>>> headers = get_headers(fname,'\\t')
|
||||
>>> is_bed(headers)
|
||||
True
|
||||
"""
|
||||
|
||||
col1_startswith = ['chr', 'chl', 'groupun', 'reftig_', 'scaffold', 'super_', 'vcho']
|
||||
|
||||
try:
|
||||
if not headers:
|
||||
return False
|
||||
for hdr in headers:
|
||||
valid_col1 = False
|
||||
if len(hdr) < 3 or len(hdr) > 12:
|
||||
return False
|
||||
for str in col1_startswith:
|
||||
if hdr[0].lower().startswith(str):
|
||||
valid_col1 = True
|
||||
break
|
||||
if valid_col1:
|
||||
try:
|
||||
map( int, [hdr[1], hdr[2]] )
|
||||
except:
|
||||
return False
|
||||
if len(hdr) > 3:
|
||||
"""
|
||||
Since all 9 of these fields are optional, it is difficult to test
|
||||
for specific column values...
|
||||
"""
|
||||
optionals = hdr[3:]
|
||||
"""
|
||||
...we can, however, test complete BED definitions fairly easily.
|
||||
"""
|
||||
if len(optionals) == 9:
|
||||
try:
|
||||
map ( int, [optionals[1], optionals[3], optionals[4], optionals[5], optionals[6]] )
|
||||
except:
|
||||
return False
|
||||
score = int(optionals[1])
|
||||
if score < 0 or score > 1000:
|
||||
return False
|
||||
if optionals[2] not in ['+', '-']:
|
||||
return False
|
||||
if int(optionals[5]) != 0:
|
||||
return False
|
||||
block_count = int(optionals[6])
|
||||
"""
|
||||
Sometime the blosck_sizes and block_starts lists end in extra commas
|
||||
"""
|
||||
block_sizes = optionals[7].rstrip(',').split(',')
|
||||
block_starts = optionals[8].rstrip(',').split(',')
|
||||
if len(block_sizes) != block_count or len(block_starts) != block_count:
|
||||
return False
|
||||
elif len(optionals) > 4 and len(optionals) < 9:
|
||||
"""
|
||||
Here it gets a bit trickier, but in this case, we can be somewhat confident
|
||||
that optionals will include a strand column
|
||||
"""
|
||||
is_valid_strand = False
|
||||
for ele in optionals:
|
||||
if ele in valid_bed_strand:
|
||||
is_valid_strand = True
|
||||
if not is_valid_strand:
|
||||
return False
|
||||
else:
|
||||
return False
|
||||
return True
|
||||
except:
|
||||
return False
|
||||
|
||||
def is_interval(headers):
|
||||
"""
|
||||
Checks for 'intervalness'
|
||||
|
||||
This is the most loosely defined format and is mostly used by galaxy itself. In general, if
|
||||
the format is_column_based, but not any of the other formats, then it must be interval.
|
||||
|
||||
>>> fname = get_test_fname('test_space.bed')
|
||||
>>> headers = get_headers(fname,'\\t')
|
||||
>>> is_interval(headers)
|
||||
False
|
||||
>>> fname = get_test_fname('interval.interval')
|
||||
>>> headers = get_headers(fname,'\\t')
|
||||
>>> is_interval(headers)
|
||||
True
|
||||
"""
|
||||
|
||||
try:
|
||||
#return is_bed(headers, skip=1) and headers[0][0][0] == '#'
|
||||
"""
|
||||
If we got here, we already know the file is_column_based and is not bed,
|
||||
so we'll just look for some valid data.
|
||||
"""
|
||||
for hdr in headers:
|
||||
if not (hdr[0] == '' or hdr[0].startswith( '#' )):
|
||||
if len(hdr) < 3:
|
||||
return False
|
||||
try:
|
||||
map( int, [hdr[1], hdr[2]] )
|
||||
except:
|
||||
return False
|
||||
return True
|
||||
except:
|
||||
return False
|
||||
|
||||
def is_html(headers):
|
||||
"""
|
||||
Determines wether the file is in html format
|
||||
|
||||
>>> headers = get_headers(__file__,' ')
|
||||
>>> is_html(headers)
|
||||
False
|
||||
>>> fname = get_test_fname('file.html')
|
||||
>>> headers = get_headers(fname,' ')
|
||||
>>> is_html(headers)
|
||||
True
|
||||
"""
|
||||
try:
|
||||
for idx, hdr in enumerate(headers):
|
||||
if hdr and hdr[0].lower() == '<html>':
|
||||
return True
|
||||
if idx > 29:
|
||||
"""
|
||||
This is a weakness since it assumes < 29 blank lines, comments, etc.
|
||||
"""
|
||||
break
|
||||
return False
|
||||
except:
|
||||
return False
|
||||
|
||||
def guess_ext(fname):
|
||||
def guess_ext( fname ):
|
||||
"""
|
||||
Returns an extension that can be used in the datatype factory to
|
||||
generate a data for the 'fname' file
|
||||
@@ -552,65 +151,46 @@ def guess_ext(fname):
|
||||
'gff'
|
||||
>>> fname = get_test_fname('gff_version_3.gff')
|
||||
>>> guess_ext(fname)
|
||||
'gff'
|
||||
'gff3'
|
||||
>>> fname = get_test_fname('temp.txt')
|
||||
>>> file(fname, 'wt').write("a 2\\nc 1")
|
||||
>>> file(fname, 'wt').write("a 2\\nc 1\\nd 0")
|
||||
>>> guess_ext(fname)
|
||||
'tabular'
|
||||
>>> fname = get_test_fname('temp.txt')
|
||||
>>> file(fname, 'wt').write("a 1 2 x\\nb 3 4 y")
|
||||
>>> file(fname, 'wt').write("a 1 2 x\\nb 3 4 y\\nc 5 6 z")
|
||||
>>> guess_ext(fname)
|
||||
'interval'
|
||||
|
||||
"""
|
||||
try:
|
||||
'tabular'
|
||||
"""
|
||||
datatypes_registry = registry.Registry()
|
||||
for datatype in datatypes_registry.sniff_order:
|
||||
"""
|
||||
The order in which we attempt to determine data format is pretty important
|
||||
because some formats are much more flexibly defined than others. Interval format
|
||||
is the most loosely defined, so it should be that last format we check.
|
||||
Some classes may not have a sniff function, which is ok. In fact, the
|
||||
Tabular and Text classes are 2 examples of classes that should never have
|
||||
a sniff function. Since these classes are default classes, they contain
|
||||
few rules to filter out data of other formats, so they should be called
|
||||
from this function after all other datatypes in sniff_order have not been
|
||||
successfully discovered.
|
||||
"""
|
||||
#guess if data is binary
|
||||
for line in file(fname):
|
||||
for char in line:
|
||||
if ord(char) > 128:
|
||||
return "data"
|
||||
|
||||
headers = get_headers(fname, None)
|
||||
if is_maf(headers):
|
||||
return 'maf'
|
||||
elif is_lav(headers):
|
||||
return 'lav'
|
||||
elif is_fasta(headers):
|
||||
return 'fasta'
|
||||
elif is_wiggle(headers):
|
||||
return 'wig'
|
||||
elif is_html(headers):
|
||||
return 'html'
|
||||
elif is_axt(headers):
|
||||
return 'axt'
|
||||
try:
|
||||
format = datatype.sniff( fname )
|
||||
except:
|
||||
format = ''
|
||||
if format:
|
||||
return format
|
||||
|
||||
if is_column_based(fname, ' ', 1):
|
||||
sep2tabs(fname)
|
||||
|
||||
if is_column_based(fname, '\t', 0):
|
||||
headers = get_headers(fname, '\t')
|
||||
if is_bed(headers):
|
||||
return 'bed'
|
||||
|
||||
if is_column_based(fname, '\t', 1):
|
||||
headers = get_headers(fname, '\t')
|
||||
if is_gff(headers):
|
||||
return 'gff'
|
||||
if is_gff3(headers):
|
||||
return 'gff3'
|
||||
elif is_interval(headers):
|
||||
return 'interval'
|
||||
"""Default binary file extension"""
|
||||
for line in file( fname ):
|
||||
for char in line:
|
||||
if ord(char) > 128:
|
||||
return "data"
|
||||
else:
|
||||
return 'tabular'
|
||||
except:
|
||||
pass
|
||||
|
||||
return 'text'
|
||||
break
|
||||
break
|
||||
if is_column_based( fname, ' ', 1 ):
|
||||
sep2tabs(fname)
|
||||
if is_column_based( fname, '\t', 1):
|
||||
return "tabular"
|
||||
return 'txt'
|
||||
|
||||
if __name__ == '__main__':
|
||||
import doctest, sys
|
||||
|
||||
+30
-101
@@ -6,7 +6,7 @@ import pkg_resources
|
||||
pkg_resources.require( "bx-python" )
|
||||
|
||||
import logging
|
||||
import data, sniff
|
||||
import data
|
||||
from galaxy import util
|
||||
from cgi import escape
|
||||
from galaxy.datatypes import metadata
|
||||
@@ -15,7 +15,6 @@ from galaxy.datatypes.metadata import MetadataAttributes
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Tabular( data.Text ):
|
||||
"""Tab delimited data"""
|
||||
|
||||
@@ -40,118 +39,47 @@ class Tabular( data.Text ):
|
||||
"""
|
||||
if dataset.has_data():
|
||||
column_types = []
|
||||
format = sniff.guess_ext( dataset.file_name )
|
||||
|
||||
if dataset.extension != format:
|
||||
"""
|
||||
We need to rely on the ability of our sniffer to properly detect datatypes here since
|
||||
there are many ways that the datatype could be improperly set.
|
||||
TODO: we may want to automatically convert the datset to the proper datatype here,
|
||||
but for now we'll leave it and just rely on the value in format.
|
||||
"""
|
||||
pass
|
||||
|
||||
"""
|
||||
The 'proceed' value allows us to skip different numbers of lines based on the data
|
||||
format (type). We need this because different formats can include lines of information
|
||||
that are not properly commented (properly commented lines start with a '#' character).
|
||||
"""
|
||||
proceed = False
|
||||
col1_startswith = ['chr', 'chl', 'groupun', 'reftig_', 'scaffold', 'super_', 'vcho']
|
||||
|
||||
for i, line in enumerate( file ( dataset.file_name ) ):
|
||||
line = line.rstrip('\r\n')
|
||||
valid = True
|
||||
if line and not line.startswith( '#' ):
|
||||
elems = line.split( '\t' )
|
||||
elems_len = len(elems)
|
||||
|
||||
if elems_len > 0:
|
||||
if format == 'bed':
|
||||
for str in col1_startswith:
|
||||
if elems[0].lower().startswith(str):
|
||||
proceed = True
|
||||
break
|
||||
elif format == 'interval':
|
||||
if elems_len > 2:
|
||||
"""Set the columns metadata attribute"""
|
||||
if elems_len != dataset.metadata.columns:
|
||||
dataset.metadata.columns = elems_len
|
||||
|
||||
"""Set the column_types metadata attribute"""
|
||||
for col in range(0, elems_len):
|
||||
col_type = None
|
||||
val = elems[col]
|
||||
if not col_type:
|
||||
"""See if val is an int"""
|
||||
try:
|
||||
map( int, [elems[1], elems[2]] )
|
||||
proceed = True
|
||||
except:
|
||||
pass # proceed is False
|
||||
elif format == 'gff':
|
||||
if elems_len == 9:
|
||||
try:
|
||||
map( int, [hdr[3], hdr[4]] )
|
||||
proceed = True
|
||||
except:
|
||||
int( val )
|
||||
col_type = 'int'
|
||||
except:
|
||||
pass
|
||||
elif format == 'gff3':
|
||||
valid_gff3_strand = ['+', '-', '.', '?']
|
||||
valid_start = False
|
||||
valid_end = False
|
||||
if elems_len == 9:
|
||||
if not col_type:
|
||||
"""See if val is a float"""
|
||||
try:
|
||||
start = int(hdr[3])
|
||||
valid_start = True
|
||||
float( val )
|
||||
col_type = 'float'
|
||||
except:
|
||||
if hdr[3] == '.':
|
||||
valid_start = True
|
||||
try:
|
||||
end = int(hdr[4])
|
||||
valid_end = True
|
||||
except:
|
||||
if hdr[4] == '.':
|
||||
valid_end = True
|
||||
srand = hdr[6]
|
||||
if valid_start and valid_end and start < end and strand in valid_gff3_strand:
|
||||
proceed = True
|
||||
elif format=='wig':
|
||||
try:
|
||||
int( elems[0] )
|
||||
proceed = True
|
||||
except:
|
||||
for str in col1_startswith:
|
||||
if elems[0].lower().startswith(str):
|
||||
proceed = True
|
||||
break
|
||||
elif ( format =='tabular' or format == "customtrack" or format == 'gbrowsetrack' ) and i > 1:
|
||||
proceed = True
|
||||
|
||||
if proceed:
|
||||
"""Set the columns metadata attribute"""
|
||||
if elems_len != dataset.metadata.columns:
|
||||
dataset.metadata.columns = elems_len
|
||||
|
||||
"""Set the column_types metadata attribute"""
|
||||
for col in range(0, elems_len):
|
||||
col_type = None
|
||||
val = elems[col]
|
||||
if not col_type:
|
||||
"""See if val is an int"""
|
||||
try:
|
||||
int( val )
|
||||
col_type = 'int'
|
||||
except:
|
||||
pass
|
||||
if not col_type:
|
||||
"""See if val is a float"""
|
||||
try:
|
||||
float( val )
|
||||
if val and val.strip().lower() == 'na':
|
||||
col_type = 'float'
|
||||
except:
|
||||
if val and val.strip().lower() == 'na':
|
||||
col_type = 'float'
|
||||
if not col_type:
|
||||
"""See if val is a list"""
|
||||
val_elems = val.split(',')
|
||||
if len( val_elems ) > 1:
|
||||
col_type = 'list'
|
||||
if not col_type:
|
||||
"""All parameters are strings, so this will be the default"""
|
||||
col_type = 'str'
|
||||
if not col_type:
|
||||
"""See if val is a list"""
|
||||
val_elems = val.split(',')
|
||||
if len( val_elems ) > 1:
|
||||
col_type = 'list'
|
||||
if not col_type:
|
||||
"""All parameters are strings, so this will be the default"""
|
||||
col_type = 'str'
|
||||
|
||||
column_types.append(col_type)
|
||||
column_types.append(col_type)
|
||||
if len(column_types) > 0: break
|
||||
|
||||
if i > 100: break # Hopefully we never get here...
|
||||
@@ -204,4 +132,5 @@ class Tabular( data.Text ):
|
||||
def before_edit( self, dataset ):
|
||||
data.Text.before_edit( self, dataset )
|
||||
if self.missing_meta( dataset ):
|
||||
self.set_meta( dataset )
|
||||
self.set_meta( dataset )
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# gff-version 2
|
||||
## Date: Thu Dec 8 19:46:27 2005
|
||||
## gff.pl $Rev: 601 $
|
||||
## Input file: /depot/data2/galaxy/encode-data/datasets/msa.AR.20051208.bed
|
||||
##gff-version 2
|
||||
##Date: Thu Dec 8 19:46:27 2005
|
||||
##gff.pl $Rev: 601 $
|
||||
##Input file: /depot/data2/galaxy/encode-data/datasets/msa.AR.20051208.bed
|
||||
|
||||
chr7 bed2gff AR 26731313 26731437 . + . score
|
||||
chr7 bed2gff AR 26731491 26731536 . + . score
|
||||
|
||||
@@ -105,7 +105,6 @@ static_style_dir = %(here)s/static/june_2007_style/blue
|
||||
data = galaxy.datatypes.data:Data,application/octet-stream
|
||||
bed = galaxy.datatypes.interval:Bed
|
||||
txt = galaxy.datatypes.data:Text
|
||||
text = galaxy.datatypes.data:Text
|
||||
interval = galaxy.datatypes.interval:Interval
|
||||
tabular = galaxy.datatypes.tabular:Tabular
|
||||
png = galaxy.datatypes.images:Image,image/png
|
||||
@@ -123,7 +122,6 @@ laj = galaxy.datatypes.images:Laj
|
||||
lav = galaxy.datatypes.sequence:Lav
|
||||
html = galaxy.datatypes.images:Html,text/html
|
||||
customtrack = galaxy.datatypes.interval:CustomTrack
|
||||
gbrowsetrack = galaxy.datatypes.interval:GBrowseTrack
|
||||
#EMBOSS TOOLS
|
||||
match = galaxy.datatypes.data:Text
|
||||
genbank = galaxy.datatypes.data:Text
|
||||
@@ -171,3 +169,21 @@ swiss = galaxy.datatypes.data:Text
|
||||
phylipnon = galaxy.datatypes.data:Text
|
||||
nexusnon = galaxy.datatypes.data:Text
|
||||
nametable = galaxy.datatypes.data:Text
|
||||
|
||||
# ---- Data Type Sniff Order --------------------------------------------------
|
||||
|
||||
[galaxy:sniff_order]
|
||||
|
||||
05 = galaxy.datatypes.images:Gmaj
|
||||
10 = galaxy.datatypes.images:Laj
|
||||
15 = galaxy.datatypes.sequence:Maf
|
||||
20 = galaxy.datatypes.sequence:Lav
|
||||
25 = galaxy.datatypes.sequence:Fasta
|
||||
30 = galaxy.datatypes.interval:Wiggle
|
||||
35 = galaxy.datatypes.images:Html
|
||||
40 = galaxy.datatypes.sequence:Axt
|
||||
45 = galaxy.datatypes.interval:Bed
|
||||
50 = galaxy.datatypes.interval:CustomTrack
|
||||
55 = galaxy.datatypes.interval:Gff
|
||||
60 = galaxy.datatypes.interval:Gff3
|
||||
65 = galaxy.datatypes.interval:Interval
|
||||
|
||||
Reference in New Issue
Block a user