diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index aa3e68f00c3..e8eed45e3ad 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -430,6 +430,7 @@ class Bed( Interval ): class Gff( Tabular ): """Tab delimited data in Gff format""" file_ext = "gff" + column_names = [ 'Seqname', 'Source', 'Feature', 'Start', 'End', 'Score', 'Strand', 'Frame', 'Group' ] """Add metadata elements""" MetadataElement( name="columns", default=9, desc="Number of columns", readonly=True, visible=False ) @@ -455,8 +456,21 @@ class Gff( Tabular ): pass Tabular.set_meta( self, dataset, skip=i ) - def make_html_table( self, dataset ): - return Tabular.make_html_table( self, dataset, skipchars=['#'] ) + def make_html_table( self, dataset, skipchars=[] ): + """Create HTML table, used for displaying peek""" + out = [''] + comments = [] + try: + # Generate column header + out.append( '' ) + for i, name in enumerate( self.column_names ): + out.append( '' % ( str( i+1 ), name ) ) + out.append( self.make_html_peek_rows( dataset, skipchars=skipchars ) ) + out.append( '
%s.%s
' ) + out = "".join( out ) + except Exception, exc: + out = "Can't create peek %s" % exc + return out def as_gbrowse_display_file( self, dataset, **kwd ): """Returns file contents that can be displayed in GBrowse apps.""" @@ -465,26 +479,40 @@ class Gff( Tabular ): def get_estimated_display_viewport( self, dataset ): """ - Return a chrom, start, stop tuple for viewing a file. There are slight differences between gff and gff version 3 + Return a chrom, start, stop tuple for viewing a file. There are slight differences between gff 2 and gff 3 formats. This function should correctly handle both... """ if dataset.has_data() and dataset.state == dataset.states.OK: try: - seqid_col = 0 - start_col = 3 - stop_col = 4 - peek = [] + seqid = None + start = None + stop = None for idx, line in enumerate( file( dataset.file_name ) ): - if line[0] != '#': - peek.append( line.split() ) - if idx > 10: - break - seqid, start, stop = peek[0][seqid_col], int( peek[0][start_col] ), int( peek[0][stop_col] ) - for p in peek[1:]: - if p[0] == seqid: - start = min( start, int( p[start_col] ) ) - stop = max( stop, int( p[stop_col] ) ) - except Exception, exc: + line = line.rstrip( '\r\n' ) + if line and line.startswith( '##sequence-region' ): # ##sequence-region IV 6000000 6030000 + elems = line.split() + seqid = elems[1] # IV + start = elems[2] # 6000000 + stop = elems[3] # 6030000 + if idx > 10: + break + if not seqid or not start or not stop: + # Perhaps the data is missing the gff3 comments fields. This is not good + # because we need to parse the entire dataset to find the stop / start + for idx, line in enumerate( file( dataset.file_name ) ): + line = line.rstrip( '\r\n' ) + if line and not line.startswith( '#' ): + elems = line.split( '\t' ) + if len( elems ) != 9: + continue # Invalid line + if not seqid: + seqid = elems[0] + # Assume all Sequence IDs are the same.Is this true for GFF3? + if not start or start < int( elems[3] ): + start = int( elems[3] ) + if not end or end > int( elems[4] ): + end = int( elems[4] ) + except: seqid, start, stop = ( '', '', '' ) return ( seqid, str( start ), str( stop ) ) else: @@ -494,12 +522,13 @@ class Gff( Tabular ): ret_val = [] if dataset.has_data: viewport_tuple = self.get_estimated_display_viewport( dataset ) - if viewport_tuple: - start = viewport_tuple[1] - stop = viewport_tuple[2] + seqid = viewport_tuple[0] + start = viewport_tuple[1] + stop = viewport_tuple[2] + if seqid and start and stop: for site_name, site_url in util.get_gbrowse_sites_by_build( dataset.dbkey ): if site_name in app.config.gbrowse_display_sites: - link = "%s?start=%s&stop=%s&ref=%s" % ( site_url, start, stop, dataset.dbkey ) + link = "%s?start=%s&stop=%s&ref=%s&dbkey=%s" % ( site_url, start, stop, seqid, dataset.dbkey ) ret_val.append( ( site_name, link ) ) return ret_val @@ -551,6 +580,7 @@ class Gff3( Gff ): file_ext = "gff3" valid_gff3_strand = ['+', '-', '.', '?'] valid_gff3_phase = ['.', '0', '1', '2'] + column_names = [ 'Seqid', 'Source', 'Type', 'Start', 'End', 'Score', 'Strand', 'Phase', 'Attributes' ] """Add metadata elements""" MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False ) diff --git a/tools/data_source/gbrowse_datasource.py b/tools/data_source/gbrowse_datasource.py new file mode 100644 index 00000000000..3879e8346e8 --- /dev/null +++ b/tools/data_source/gbrowse_datasource.py @@ -0,0 +1,53 @@ +#!/usr/bin/env python +#Retreives data from GMOD and stores in a file. GBrowse parameters are provided in the input/output file. +import urllib, sys, os, gzip, tempfile, shutil +from galaxy import eggs +from galaxy.datatypes import data + +assert sys.version_info[:2] >= ( 2, 4 ) + +def stop_err( msg ): + sys.stderr.write( msg ) + sys.exit() + +def __main__(): + filename = sys.argv[1] + params = {} + + for line in open( filename, 'r' ): + try: + line = line.strip() + fields = line.split( '\t' ) + params[ fields[0] ] = fields[1] + except: + continue + + URL = params.get( 'URL', None ) + if not URL: + open( filename, 'w' ).write( "" ) + stop_err( 'Datasource has not sent back a URL parameter.' ) + + for i, param in enumerate( params.keys() ): + if i == 0: + sep = '?' + else: + sep = '&' + if param != '__collected_datasets__': + URL += "%s%s=%s" % ( sep, param, params.get( param ) ) + + CHUNK_SIZE = 2**20 # 1Mb + try: + page = urllib.urlopen( URL ) + except Exception, exc: + raise Exception( 'Problems connecting to %s (%s)' % ( URL, exc ) ) + sys.exit( 1 ) + + fp = open( filename, 'wb' ) + while 1: + chunk = page.read( CHUNK_SIZE ) + if not chunk: + break + fp.write( chunk ) + fp.close() + +if __name__ == "__main__": __main__() diff --git a/tools/data_source/gbrowse_elegans.xml b/tools/data_source/gbrowse_elegans.xml index 9bb3aa62f2a..5d25ec4621a 100644 --- a/tools/data_source/gbrowse_elegans.xml +++ b/tools/data_source/gbrowse_elegans.xml @@ -1,13 +1,13 @@ server - + gbrowse_datasource.py $output go to C. Elegans server $GALAXY_URL - + diff --git a/tools/data_source/gbrowse_filter.py b/tools/data_source/gbrowse_filter.py deleted file mode 100644 index 88363837e8c..00000000000 --- a/tools/data_source/gbrowse_filter.py +++ /dev/null @@ -1,56 +0,0 @@ -import urllib - -from galaxy import datatypes, config -from galaxy.datatypes import sniff -import tempfile, shutil - -import logging -log = logging.getLogger( __name__ ) - -def exec_before_job( app, inp_data, out_data, param_dict, tool=None ): - """Sets the name of the data""" - data_name = urllib.unquote( param_dict.get( 't', 'GBrowse query' ) ).replace( '+', ' ' ) - data_region = param_dict.get( 'q', '' ) - data_type = param_dict.get( 'type', 'txt' ) - name, data = out_data.items()[0] - if data_type == 'txt': - data_type = sniff.guess_ext( data.file_name ) - data = app.datatypes_registry.change_datatype( data, data_type ) - data.name = '%s %s' % ( data_name, data_region ) - out_data[name] = data - -def exec_after_process( app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None ): - """Verifies the data after the run""" - URL = param_dict.pop( 'URL', None ) - if not URL: - raise Exception( 'Datasource has not sent back a URL parameter' ) - for i, param in enumerate( param_dict.keys() ): - if i == 0: - sep = '?' - else: - sep = '&' - if param != '__collected_datasets__': - URL += "%s%s=%s" %( sep, param, param_dict.get( param ) ) - - CHUNK_SIZE = 2**20 # 1Mb - try: - page = urllib.urlopen( URL ) - except Exception, exc: - raise Exception( 'Problems connecting to %s (%s)' % ( URL, exc ) ) - sys.exit( 1 ) - - name, data = out_data.items()[0] - - fp = open( data.file_name, 'wb' ) - while 1: - chunk = page.read( CHUNK_SIZE ) - if not chunk: - break - fp.write( chunk ) - fp.close() - - data.info = data.name - data_type = sniff.guess_ext( data.file_name ) - data = app.datatypes_registry.change_datatype( data, data_type ) - data.set_peek() - data.flush() diff --git a/tools/data_source/gbrowse_filter_code.py b/tools/data_source/gbrowse_filter_code.py new file mode 100644 index 00000000000..9f6b95506bd --- /dev/null +++ b/tools/data_source/gbrowse_filter_code.py @@ -0,0 +1,34 @@ +# Code for direct connection to GMOD +from galaxy.datatypes import sniff +import urllib + +import logging +log = logging.getLogger( __name__ ) + +def exec_before_job( app, inp_data, out_data, param_dict, tool=None ): + """Sets the attributes of the data""" + gb_settings = urllib.unquote( param_dict.get( 't', None ) ) # t=CG+TS+ESTB+SAGE+EXPR+EXPR_PATTERN+SNPs+PolyA+BLASTX+LINK+ETILE + gb_landmark_region = urllib.unquote( param_dict.get( 'q' ) ) # q=IV:6070000..6100000& + gb_land_mark, gb_region = gb_landmark_region.split( ':' ) + items = out_data.items() + for name, data in items: + data.name = "%s on %s" % ( data.name, gb_landmark_region ) + data.dbkey = param_dict.get( 'dbkey', '?' ) + # Store GMOD / GBrowse parameters temporarily in output file + out = open( data.file_name, 'w' ) + for key, value in param_dict.items(): + out.write( "%s\t%s\n" % ( key, value ) ) + out.close() + out_data[ name ] = data + +def exec_after_process( app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None ): + """Verifies the data after the run""" + name, data = out_data.items()[0] + if data.state == data.states.OK: + data.info = data.name + if data.extension == 'txt': + data_type = sniff.guess_ext( data.file_name ) + data = app.datatypes_registry.change_datatype( data, data_type ) + data.set_peek() + data.set_size() + data.flush()