More fixes for GBrowse / GMOD communication, tool now includes command line, display in GBrowse links now functional.

This commit is contained in:
Greg Von Kuster
2008-05-15 18:46:51 +00:00
parent 76e128a6ee
commit c424782fb4
5 changed files with 140 additions and 79 deletions
+51 -21
View File
@@ -430,6 +430,7 @@ class Bed( Interval ):
class Gff( Tabular ):
"""Tab delimited data in Gff format"""
file_ext = "gff"
column_names = [ 'Seqname', 'Source', 'Feature', 'Start', 'End', 'Score', 'Strand', 'Frame', 'Group' ]
"""Add metadata elements"""
MetadataElement( name="columns", default=9, desc="Number of columns", readonly=True, visible=False )
@@ -455,8 +456,21 @@ class Gff( Tabular ):
pass
Tabular.set_meta( self, dataset, skip=i )
def make_html_table( self, dataset ):
return Tabular.make_html_table( self, dataset, skipchars=['#'] )
def make_html_table( self, dataset, skipchars=[] ):
"""Create HTML table, used for displaying peek"""
out = ['<table cellspacing="0" cellpadding="3">']
comments = []
try:
# Generate column header
out.append( '<tr>' )
for i, name in enumerate( self.column_names ):
out.append( '<th>%s.%s</th>' % ( str( i+1 ), name ) )
out.append( self.make_html_peek_rows( dataset, skipchars=skipchars ) )
out.append( '</table>' )
out = "".join( out )
except Exception, exc:
out = "Can't create peek %s" % exc
return out
def as_gbrowse_display_file( self, dataset, **kwd ):
"""Returns file contents that can be displayed in GBrowse apps."""
@@ -465,26 +479,40 @@ class Gff( Tabular ):
def get_estimated_display_viewport( self, dataset ):
"""
Return a chrom, start, stop tuple for viewing a file. There are slight differences between gff and gff version 3
Return a chrom, start, stop tuple for viewing a file. There are slight differences between gff 2 and gff 3
formats. This function should correctly handle both...
"""
if dataset.has_data() and dataset.state == dataset.states.OK:
try:
seqid_col = 0
start_col = 3
stop_col = 4
peek = []
seqid = None
start = None
stop = None
for idx, line in enumerate( file( dataset.file_name ) ):
if line[0] != '#':
peek.append( line.split() )
if idx > 10:
break
seqid, start, stop = peek[0][seqid_col], int( peek[0][start_col] ), int( peek[0][stop_col] )
for p in peek[1:]:
if p[0] == seqid:
start = min( start, int( p[start_col] ) )
stop = max( stop, int( p[stop_col] ) )
except Exception, exc:
line = line.rstrip( '\r\n' )
if line and line.startswith( '##sequence-region' ): # ##sequence-region IV 6000000 6030000
elems = line.split()
seqid = elems[1] # IV
start = elems[2] # 6000000
stop = elems[3] # 6030000
if idx > 10:
break
if not seqid or not start or not stop:
# Perhaps the data is missing the gff3 comments fields. This is not good
# because we need to parse the entire dataset to find the stop / start
for idx, line in enumerate( file( dataset.file_name ) ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ):
elems = line.split( '\t' )
if len( elems ) != 9:
continue # Invalid line
if not seqid:
seqid = elems[0]
# Assume all Sequence IDs are the same.Is this true for GFF3?
if not start or start < int( elems[3] ):
start = int( elems[3] )
if not end or end > int( elems[4] ):
end = int( elems[4] )
except:
seqid, start, stop = ( '', '', '' )
return ( seqid, str( start ), str( stop ) )
else:
@@ -494,12 +522,13 @@ class Gff( Tabular ):
ret_val = []
if dataset.has_data:
viewport_tuple = self.get_estimated_display_viewport( dataset )
if viewport_tuple:
start = viewport_tuple[1]
stop = viewport_tuple[2]
seqid = viewport_tuple[0]
start = viewport_tuple[1]
stop = viewport_tuple[2]
if seqid and start and stop:
for site_name, site_url in util.get_gbrowse_sites_by_build( dataset.dbkey ):
if site_name in app.config.gbrowse_display_sites:
link = "%s?start=%s&stop=%s&ref=%s" % ( site_url, start, stop, dataset.dbkey )
link = "%s?start=%s&stop=%s&ref=%s&dbkey=%s" % ( site_url, start, stop, seqid, dataset.dbkey )
ret_val.append( ( site_name, link ) )
return ret_val
@@ -551,6 +580,7 @@ class Gff3( Gff ):
file_ext = "gff3"
valid_gff3_strand = ['+', '-', '.', '?']
valid_gff3_phase = ['.', '0', '1', '2']
column_names = [ 'Seqid', 'Source', 'Type', 'Start', 'End', 'Score', 'Strand', 'Phase', 'Attributes' ]
"""Add metadata elements"""
MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False )
+53
View File
@@ -0,0 +1,53 @@
#!/usr/bin/env python
#Retreives data from GMOD and stores in a file. GBrowse parameters are provided in the input/output file.
import urllib, sys, os, gzip, tempfile, shutil
from galaxy import eggs
from galaxy.datatypes import data
assert sys.version_info[:2] >= ( 2, 4 )
def stop_err( msg ):
sys.stderr.write( msg )
sys.exit()
def __main__():
filename = sys.argv[1]
params = {}
for line in open( filename, 'r' ):
try:
line = line.strip()
fields = line.split( '\t' )
params[ fields[0] ] = fields[1]
except:
continue
URL = params.get( 'URL', None )
if not URL:
open( filename, 'w' ).write( "" )
stop_err( 'Datasource has not sent back a URL parameter.' )
for i, param in enumerate( params.keys() ):
if i == 0:
sep = '?'
else:
sep = '&'
if param != '__collected_datasets__':
URL += "%s%s=%s" % ( sep, param, params.get( param ) )
CHUNK_SIZE = 2**20 # 1Mb
try:
page = urllib.urlopen( URL )
except Exception, exc:
raise Exception( 'Problems connecting to %s (%s)' % ( URL, exc ) )
sys.exit( 1 )
fp = open( filename, 'wb' )
while 1:
chunk = page.read( CHUNK_SIZE )
if not chunk:
break
fp.write( chunk )
fp.close()
if __name__ == "__main__": __main__()
+2 -2
View File
@@ -1,13 +1,13 @@
<?xml version="1.0"?>
<tool name="C. Elegans" id="gbrowse_elegans">
<description>server</description>
<command/>
<command interpreter="python">gbrowse_datasource.py $output</command>
<inputs action="http://brie3.cshl.edu:9000/cgi-bin/gbrowse/elegans/" check_values="false" method="get" target="_top">
<display>go to C. Elegans server $GALAXY_URL</display>
<param name="GALAXY_URL" type="baseurl" value="/tool_runner/gbrowse_elegans" />
</inputs>
<uihints minwidth="800"/>
<code file="gbrowse_filter.py"/>
<code file="gbrowse_filter_code.py"/>
<outputs>
<data name="output" format="txt" />
</outputs>
-56
View File
@@ -1,56 +0,0 @@
import urllib
from galaxy import datatypes, config
from galaxy.datatypes import sniff
import tempfile, shutil
import logging
log = logging.getLogger( __name__ )
def exec_before_job( app, inp_data, out_data, param_dict, tool=None ):
"""Sets the name of the data"""
data_name = urllib.unquote( param_dict.get( 't', 'GBrowse query' ) ).replace( '+', ' ' )
data_region = param_dict.get( 'q', '' )
data_type = param_dict.get( 'type', 'txt' )
name, data = out_data.items()[0]
if data_type == 'txt':
data_type = sniff.guess_ext( data.file_name )
data = app.datatypes_registry.change_datatype( data, data_type )
data.name = '%s %s' % ( data_name, data_region )
out_data[name] = data
def exec_after_process( app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None ):
"""Verifies the data after the run"""
URL = param_dict.pop( 'URL', None )
if not URL:
raise Exception( 'Datasource has not sent back a URL parameter' )
for i, param in enumerate( param_dict.keys() ):
if i == 0:
sep = '?'
else:
sep = '&'
if param != '__collected_datasets__':
URL += "%s%s=%s" %( sep, param, param_dict.get( param ) )
CHUNK_SIZE = 2**20 # 1Mb
try:
page = urllib.urlopen( URL )
except Exception, exc:
raise Exception( 'Problems connecting to %s (%s)' % ( URL, exc ) )
sys.exit( 1 )
name, data = out_data.items()[0]
fp = open( data.file_name, 'wb' )
while 1:
chunk = page.read( CHUNK_SIZE )
if not chunk:
break
fp.write( chunk )
fp.close()
data.info = data.name
data_type = sniff.guess_ext( data.file_name )
data = app.datatypes_registry.change_datatype( data, data_type )
data.set_peek()
data.flush()
+34
View File
@@ -0,0 +1,34 @@
# Code for direct connection to GMOD
from galaxy.datatypes import sniff
import urllib
import logging
log = logging.getLogger( __name__ )
def exec_before_job( app, inp_data, out_data, param_dict, tool=None ):
"""Sets the attributes of the data"""
gb_settings = urllib.unquote( param_dict.get( 't', None ) ) # t=CG+TS+ESTB+SAGE+EXPR+EXPR_PATTERN+SNPs+PolyA+BLASTX+LINK+ETILE
gb_landmark_region = urllib.unquote( param_dict.get( 'q' ) ) # q=IV:6070000..6100000&
gb_land_mark, gb_region = gb_landmark_region.split( ':' )
items = out_data.items()
for name, data in items:
data.name = "%s on %s" % ( data.name, gb_landmark_region )
data.dbkey = param_dict.get( 'dbkey', '?' )
# Store GMOD / GBrowse parameters temporarily in output file
out = open( data.file_name, 'w' )
for key, value in param_dict.items():
out.write( "%s\t%s\n" % ( key, value ) )
out.close()
out_data[ name ] = data
def exec_after_process( app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None ):
"""Verifies the data after the run"""
name, data = out_data.items()[0]
if data.state == data.states.OK:
data.info = data.name
if data.extension == 'txt':
data_type = sniff.guess_ext( data.file_name )
data = app.datatypes_registry.change_datatype( data, data_type )
data.set_peek()
data.set_size()
data.flush()