From 6a0aa166aa8a1691071eed96f248e593da5c15fc Mon Sep 17 00:00:00 2001 From: Ian Schenck Date: Mon, 19 Jan 2009 17:45:28 -0500 Subject: [PATCH] Initial GeneTrack commit. Most parts are in, but dependencies will still be a problem. --- datatypes_conf.xml.sample | 2 + lib/galaxy/datatypes/coverage.py | 30 +++++ lib/galaxy/datatypes/registry.py | 6 +- lib/galaxy/datatypes/tracks.py | 30 +++++ lib/galaxy/web/controllers/genetrack.py | 161 ++++++++++++++++++++++++ scripts/paster.py | 2 +- static/genetrack/genetrack.css | 78 ++++++++++++ static/genetrack/genetrack.js | 79 ++++++++++++ tool_conf.xml.sample | 4 +- tools/sr_mapping/lastz_wrapper.xml | 2 +- tools/visualization/genetrack.py | 139 ++++++++++++++++++++ tools/visualization/genetrack.xml | 53 ++++++++ tools/visualization/genetrack_code.py | 13 ++ 13 files changed, 594 insertions(+), 5 deletions(-) create mode 100644 lib/galaxy/datatypes/coverage.py create mode 100644 lib/galaxy/datatypes/tracks.py create mode 100644 lib/galaxy/web/controllers/genetrack.py create mode 100644 static/genetrack/genetrack.css create mode 100644 static/genetrack/genetrack.js create mode 100644 tools/visualization/genetrack.py create mode 100644 tools/visualization/genetrack.xml create mode 100644 tools/visualization/genetrack_code.py diff --git a/datatypes_conf.xml.sample b/datatypes_conf.xml.sample index 431d9da7c95..391f7937663 100644 --- a/datatypes_conf.xml.sample +++ b/datatypes_conf.xml.sample @@ -82,6 +82,8 @@ + + diff --git a/lib/galaxy/datatypes/coverage.py b/lib/galaxy/datatypes/coverage.py new file mode 100644 index 00000000000..4bd76425a71 --- /dev/null +++ b/lib/galaxy/datatypes/coverage.py @@ -0,0 +1,30 @@ +""" +Coverage datatypes + +""" +import pkg_resources +pkg_resources.require( "bx-python" ) + +import logging, os, sys, time, sets, tempfile, shutil +import data +from galaxy import util +from galaxy.datatypes.sniff import * +from galaxy.web import url_for +from cgi import escape +import urllib +from bx.intervals.io import * +from galaxy.datatypes import metadata +from galaxy.datatypes.metadata import MetadataElement +from galaxy.datatypes.tabular import Tabular + +log = logging.getLogger(__name__) + +class LastzCoverage( Tabular ): + file_ext = "coverage" + + MetadataElement( name="chromCol", default=1, desc="Chrom column", param=metadata.ColumnParameter ) + MetadataElement( name="positionCol", default=2, desc="Position column", param=metadata.ColumnParameter ) + MetadataElement( name="forwardCol", default=3, desc="Forward or aggregate read column", param=metadata.ColumnParameter ) + MetadataElement( name="reverseCol", desc="Optional reverse read column", param=metadata.ColumnParameter, optional=True, no_value=0 ) + MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True, visible=False ) + \ No newline at end of file diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index c27cce91188..1639e904964 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -3,7 +3,7 @@ Provides mapping between extensions and datatypes, mime-types, etc. """ import os import logging -import data, tabular, interval, images, sequence, qualityscore, genetics, xml +import data, tabular, interval, images, sequence, qualityscore, genetics, xml, coverage, tracks import galaxy.util from galaxy.util.odict import odict @@ -97,12 +97,14 @@ class Registry( object ): 'bed' : interval.Bed(), 'binseq.zip' : images.Binseq(), 'blastxml' : xml.BlastXml(), + 'coverage' : coverage.LastzCoverage(), 'customtrack' : interval.CustomTrack(), 'csfasta' : sequence.csFasta(), 'fasta' : sequence.Fasta(), 'fastqsolexa' : sequence.FastqSolexa(), 'gff' : interval.Gff(), - 'gff3' : interval.Gff3(), + 'gff3' : interval.Gff3(), + 'genetrack' : tracks.GeneTrack(), 'interval' : interval.Interval(), 'laj' : images.Laj(), 'lav' : sequence.Lav(), diff --git a/lib/galaxy/datatypes/tracks.py b/lib/galaxy/datatypes/tracks.py new file mode 100644 index 00000000000..1c5a9291bef --- /dev/null +++ b/lib/galaxy/datatypes/tracks.py @@ -0,0 +1,30 @@ +""" +Datatype classes for tracks/track views within galaxy. +""" + +import data +import logging +import re +from cgi import escape +from galaxy.datatypes.metadata import MetadataElement +from galaxy.datatypes import metadata +import galaxy.model +from galaxy import util +from galaxy.web import url_for +from sniff import * + +log = logging.getLogger(__name__) + +class GeneTrack( data.Binary ): + file_ext = "genetrack" + + MetadataElement( name="hdf", default="data.hdf", desc="HDF DB", readonly=True, visible=True, no_value=0 ) + MetadataElement( name="sqlite", default="features.sqlite", desc="SQLite Features DB", readonly=True, visible=True, no_value=0 ) + MetadataElement( name="label", default="Custom", desc="Track Label", readonly=True, visible=True, no_value="Custom" ) + + def __init__(self, **kwargs): + super(GeneTrack, self).__init__(**kwargs) + self.add_display_app( 'genetrack', 'View in ', '', 'genetrack_link' ) + + def genetrack_link( self, dataset, type, app, base_url ): + return [('GeneTrack', url_for(controller='genetrack', action='index', dataset_id=dataset.id ))] \ No newline at end of file diff --git a/lib/galaxy/web/controllers/genetrack.py b/lib/galaxy/web/controllers/genetrack.py new file mode 100644 index 00000000000..ae81a277cd4 --- /dev/null +++ b/lib/galaxy/web/controllers/genetrack.py @@ -0,0 +1,161 @@ +import time, glob, os + +import pkg_resources +pkg_resources.require("GeneTrack") + +import atlas +from atlas import sql +from atlas import util as atlas_utils +from atlas.web import formlib +from mako import exceptions +from mako.template import Template +from mako.lookup import TemplateLookup +from galaxy.web.base.controller import * + +pkg_resources.require( "Paste" ) +import paste.httpexceptions + +# SETUP Track Builders +from mod454.trackbuilder import build_tracks +import functools +def twostrand_tracks( param=None, conf=None ): + return build_tracks( data_label=conf.LABEL, fit_label=conf.FIT_LABEL, pred_label=conf.PRED_LABEL, param=param, conf=conf, strand='twostrand') +def composite_tracks( param=None, conf=None ): + return build_tracks( data_label=conf.LABEL, fit_label=conf.FIT_LABEL, pred_label=conf.PRED_LABEL, param=param, conf=conf, strand='composite') + +class BaseConf( object ): + """ + Fake web_conf for atlas. + """ + IMAGE_DIR = "static/genetrack/plots/" + LEVELS = [str(x) for x in [ 50, 100, 250, 500, 1000, 2500, 5000, 10000, 20000, 50000, 100000, 200000 ]] + ZOOM_LEVELS = zip(LEVELS, LEVELS) + PLOT_SETUP = [ + ('comp-id', 'Composite' , 'genetrack/index.html', composite_tracks ), + ('two-id' , 'Two Strand', 'genetrack/index.html', twostrand_tracks ), + ] + PLOT_CHOICES = [ (id, name) for (id, name, page, func) in PLOT_SETUP ] + PLOT_MAPPER = dict( [ (id, (page, func)) for (id, name, page, func) in PLOT_SETUP ] ) + + def __init__(self, **kwds): + for key,value in kwds.items(): + setattr( self, key, value) + +class WebRoot(BaseController): + @web.expose + def search(self, trans, word='', dataset_id=None, submit=''): + """ + Default search page + """ + data = trans.app.model.HistoryDatasetAssociation.get( dataset_id ) + if not data: + raise paste.httpexceptions.HTTPRequestRangeNotSatisfiable( "Invalid reference dataset id: %s." % str( dataset_id ) ) + # the main configuration file + conf = BaseConf( + TITLE = "%s: %s" % (data.metadata.dbkey, data.metadata.label), + HDF_DATABASE = os.path.join( data.extra_files_path, data.metadata.hdf ), + SQL_URI = "sqlite:///%s" % os.path.join( data.extra_files_path, data.metadata.sqlite ), + LABEL = data.metadata.label, + FIT_LABEL = "%s-SIGMA-%d" % (data.metadata.label, 20), + PRED_LABEL = "PRED-%s-SIGMA-%d" % (data.metadata.label, 20), + ) + from atlas import hdf + db = hdf.hdf_open( conf.HDF_DATABASE, mode='r' ) + conf.CHROM_FIELDS = [(x,x) for x in hdf.GroupData(db=db, name=conf.LABEL).labels] + db.close() + + param = atlas.Param( word=word ) + # search with features based on param.feature + + # search for a given + session = sql.get_session( conf.SQL_URI ) + + if param.word: + def search_query( word, text ): + query = session.query(sql.Feature).filter( "name LIKE :word or freetext LIKE :text" ).params(word=word, text=text) + query = list(query[:20]) + return query + + # a little heuristics to match most likely target + targets = [ + (param.word+'%', 'No match'), # match beginning + ('%'+param.word+'%', 'No match'), # match name anywhere + ('%'+param.word+'%', '%'+param.word+'%'), # match json anywhere + ] + for word, text in targets: + query = search_query( word=word, text=text) + if query: + break + else: + query = [] + + return trans.fill_template_mako('genetrack/search.html', param=param, query=query, dataset_id=dataset_id) + + @web.expose + def index(self, trans, dataset_id=None, **kwds): + """ + Main request handler + """ + data = trans.app.model.HistoryDatasetAssociation.get( dataset_id ) + if not data: + raise paste.httpexceptions.HTTPRequestRangeNotSatisfiable( "Invalid reference dataset id: %s." % str( dataset_id ) ) + # the main configuration file + conf = BaseConf( + TITLE = "%s: %s" % (data.metadata.dbkey, data.metadata.label), + HDF_DATABASE = os.path.join( data.extra_files_path, data.metadata.hdf ), + SQL_URI = "sqlite:///%s" % os.path.join( data.extra_files_path, data.metadata.sqlite ), + LABEL = data.metadata.label, + FIT_LABEL = "%s-SIGMA-%d" % (data.metadata.label, 20), + PRED_LABEL = "PRED-%s-SIGMA-%d" % (data.metadata.label, 20), + ) + from atlas import hdf + db = hdf.hdf_open( conf.HDF_DATABASE, mode='r' ) + conf.CHROM_FIELDS = [(x,x) for x in hdf.GroupData(db=db, name=conf.LABEL).labels] + db.close() + + # generate a new form based on the configuration + form = formlib.main_form( conf ) + + # clear the tempdir every once in a while + atlas_utils.clear_tempdir( dir=conf.IMAGE_DIR, days=1, chance=10) + + incoming = form.defaults() + incoming.update( kwds ) + + # manage the zoom and pan requests + incoming = formlib.zoom_change( kdict=incoming, levels=conf.LEVELS) + incoming = formlib.pan_view( kdict=incoming ) + + # process the form + param = atlas.Param( **incoming ) + form.process( incoming ) + + if kwds and form.isSuccessful(): + # adds the sucessfull parameters + param.update( form.values() ) + + # if it was a search word not a number go to search page + try: + center = int( param.feature ) + except ValueError: + # go and search for these + return trans.response.send_redirect( web.url_for( controller='genetrack', action='search', word=param.feature, dataset_id=dataset_id ) ) + + # keep image at a sane size + param.width = min( [2000, int(param.img_size)] ) + + # get the template and the function used to generate the tracks + tmpl_name, track_maker = conf.PLOT_MAPPER[param.plot] + + if track_maker is not None: + # generate the name that the image will be stored at + fname, fpath = atlas_utils.make_tempfile( dir=conf.IMAGE_DIR, suffix='.png') + param.fname = fname + + # generate the track + track_chart = track_maker( param=param, conf=conf ) + track_chart.save(fname=fpath) + + return trans.fill_template_mako(tmpl_name, conf=conf, form=form, param=param, dataset_id=dataset_id) + + diff --git a/scripts/paster.py b/scripts/paster.py index 7624257a665..446cb9de90b 100755 --- a/scripts/paster.py +++ b/scripts/paster.py @@ -12,7 +12,7 @@ assert sys.version_info[:2] >= ( 2, 4 ) new_path = [ os.path.join( os.getcwd(), "lib" ) ] new_path.extend( sys.path[1:] ) # remove scripts/ from the path sys.path = new_path - +print sys.path from galaxy import eggs import pkg_resources diff --git a/static/genetrack/genetrack.css b/static/genetrack/genetrack.css new file mode 100644 index 00000000000..668300be06d --- /dev/null +++ b/static/genetrack/genetrack.css @@ -0,0 +1,78 @@ + +body { + font-family: "Trebuchet MS", Arial, tahoma, sans-serif; + font-size: 14px; + line-height: 1.6em; + margin: 0; + padding: 0; + border-top: 9px solid #CCD9FF; +} + +/* Error message style */ +.error{ + background: #FFFF66; +} + +/* Error message style */ +.message{ + background: #33FF66; +} + +/* Odd data row in the table */ +.selected { + background-color: #FFFFCC; +} + +.nav_button{ + background-color:#EEEEEE; + border:1px solid; + color: #000000; +} + +.nav_button:hover{ + background-color:#000000; + border:1px solid; + color: #FFFFFF; +} + +.grey { + background-color: #EFEFEF; +} + +.odd { + background-color: #ECECEC; +} + +.even { + background-color: #FFFFFF; +} + +/* Text table style */ +.data_table { + border: 1px solid #CCCCCC; + background-color: white; +} + +/* Footer is added to every page */ +#footer { + background: #EFEFEF; + text-align:center; + padding:.2em; + border-top: 1px solid #CCD9FF; + border-bottom: 1px solid #CCD9FF; + clear: both; +} + +#footer p { + font-size:.94em; line-height:2em; color:#cccccc; margin: 0; + } + +#tag { + font-size:.80em; margin: 4px; padding: 2px; + } + + +#footer img { + vertical-align: middle; margin-left: 3px; padding-bottom: 2px; +} + diff --git a/static/genetrack/genetrack.js b/static/genetrack/genetrack.js new file mode 100644 index 00000000000..be88d107ab8 --- /dev/null +++ b/static/genetrack/genetrack.js @@ -0,0 +1,79 @@ +var cookie_name = "genetrack_ui" +var now = new Date(); +now.setTime(now.getTime() + 365 * 24 * 60 * 60 * 1000); + +// this toggles between none and block +function toggle(name){ + var elem = get(name) + if (elem) { + if (elem.style.display=="none"){ + elem.style.display="block" + setCookie(cookie_name, name, now) + } else { + elem.style.display="none" + setCookie(cookie_name, '', now) + } + + } +} + +function main(){ + //executed upon main body load + var value = getCookie(cookie_name); + toggle( value ) +} + +// this toggles between visible and hidden +function show(name){ + var elem = get(name) + if (elem.style.visibility=="hidden"){ + elem.style.visibility="visible"; + } else { + elem.style.visibility="hidden"; + } +} + +// utility function to get the length of on object +function len(obj){ + return obj.length; +} + +// utility function to get an element by id +function get(name){ + return document.getElementById(name); +} + +// pops up a window +function pop_up(url) { + day = new Date(); + id = day.getTime(); + eval("page" + id + " = window.open(url, '" + id + "', 'toolbar=0,scrollbars=1,location=0,statusbar=1,menubar=0,resizable=1,width=500,height=300');"); +} + +// +// cookie management off the web +// http://www.webreference.com/js/column8/property.html +// +function setCookie(name, value, expires, path, domain, secure) { + var curCookie = name + "=" + escape(value) + + ((expires) ? "; expires=" + expires.toGMTString() : "") + + ((path) ? "; path=" + path : "") + + ((domain) ? "; domain=" + domain : "") + + ((secure) ? "; secure" : ""); + document.cookie = curCookie; +} + +function getCookie(name) { + var dc = document.cookie; + var prefix = name + "="; + var begin = dc.indexOf("; " + prefix); + if (begin == -1) { + begin = dc.indexOf(prefix); + if (begin != 0) return null; + } else + begin += 2; + var end = document.cookie.indexOf(";", begin); + if (end == -1) + end = dc.length; + return unescape(dc.substring(begin + prefix.length, end)); +} diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample index f46c9cef558..47cf0c1e734 100644 --- a/tool_conf.xml.sample +++ b/tool_conf.xml.sample @@ -11,7 +11,6 @@ - @@ -302,4 +301,7 @@ +
+ +
diff --git a/tools/sr_mapping/lastz_wrapper.xml b/tools/sr_mapping/lastz_wrapper.xml index e3aabf57243..f29d5fb7168 100644 --- a/tools/sr_mapping/lastz_wrapper.xml +++ b/tools/sr_mapping/lastz_wrapper.xml @@ -82,7 +82,7 @@ - + lastz diff --git a/tools/visualization/genetrack.py b/tools/visualization/genetrack.py new file mode 100644 index 00000000000..bd27d5df4d2 --- /dev/null +++ b/tools/visualization/genetrack.py @@ -0,0 +1,139 @@ +#!/usr/bin/env python +""" +Run GeneTrack(atlas) with a faked conf file to generate GeneTrack data files. + +usage: %prog + -l, --label=N: Data label for fit curve/peak plot + -1, --fits=N/N/N/N/N,...: Data files (interval format) for fit curve/peak plot + -2, --feats=N:M/N/N/N/N/N,...: Data files (interval format) for features. + -d, --data=N: Output path for hdf5 and sqlite databases. + -o, --output=N: Output path for export file. +""" +from galaxy import eggs +import pkg_resources +pkg_resources.require("GeneTrack") +pkg_resources.require("bx-python") + +from atlas import commands +from bx.cookbook import doc_optparse +import os +import commands as oscommands +import tempfile + +SIGMA = 20 +WIDTH = 5 * SIGMA +EXCLUSION_ZONE = 147 + +def main(label, fit, feats, data_dir, output): + os.mkdir(data_dir) + conf = DummyConf( + __name__=label, + CLOBBER = True, + DATA_SIZE = 3*10**6, + MINIMUM_PEAK_SIZE = 0.1, + LOADER_ENABLED = True, + FITTER_ENABLED = True, + PREDICTOR_ENABLED = True, + EXPORTER_ENABLED = True, + LOADER = loader, + FITTER = fitter, + PREDICTOR = predictor, + EXPORTER = exporter, + HDF_DATABASE = os.path.join( data_dir, "data.hdf" ), + SQL_URI = "sqlite:///%s" % os.path.join( data_dir, "features.sqlite" ), + SIGMA = SIGMA, + WIDTH = WIDTH, + DATA_LABEL = label, + FIT_LABEL = "%s-SIGMA-%d" % ( label,SIGMA ), + PEAK_LABEL = "PRED-%s-SIGMA-%d" % ( label,SIGMA ), + EXCLUSION_ZONE = EXCLUSION_ZONE, + LEFT_SHIFT = EXCLUSION_ZONE / 2, + RIGHT_SHIFT = EXCLUSION_ZONE / 2, + EXPORT_LABELS = [ "PRED-%s-SIGMA-%d" % ( label,SIGMA ) ], + EXPORT_DIR = os.path.join( data_dir ), + DATA_FILE=fit[1], + fit=fit, + feats=feats, + ) + commands.execute(conf) + +# mod454 seems to be a module without a package. The necessary funcitons are +# stubbed out here until I'm sure of their final home. INS + +def loader( conf ): + from atlas import hdf + from mod454.schema import Mod454Schema as Schema + last_chrom = table = None + db = hdf.hdf_open( conf.HDF_DATABASE, mode='a', title='HDF database') + gp = hdf.create_group( db=db, name=conf.DATA_LABEL, desc='data group', clobber=conf.CLOBBER ) + fit_meta = conf.fit[2] + # iterate over the file and insert into table + for line in open( conf.fit[1], "r" ): + if line.startswith("chrom"): continue #Skip possible header + if line.startswith("#"): continue + fields = line.rstrip('\r\n').split('\t') + chrom = fields[fit_meta.chromCol] + if chrom != last_chrom: + if table: table.flush() + table = hdf.create_table( db=db, name=chrom, where=gp, schema=Schema, clobber=False ) + last_chrom = chrom + try: + position = int(fields[fit_meta.positionCol]) + forward = float(fields[fit_meta.forwardCol]) + reverse = fit_meta.reverseCol > -1 and float(fields[fit_meta.reverseCol]) or 0.0 + row = ( position, forward, reverse, forward+reverse, ) + table.append( [ row ] ) + except ValueError: + # Ignore bad lines + pass + table.flush() + db.close() + +def fitter( conf ): + from mod454.fitter import fitter as mod454_fitter + return mod454_fitter( conf ) + +def predictor( conf ): + from mod454.predictor import predictor as mod454_predictor + return mod454_predictor( conf ) + +def exporter( conf ): + return commands.bed_exporter(conf) + +class Bunch( object ): + def __init__(self, **kwargs): + for key,value in kwargs.items(): + setattr( self, key, value ) + +class DummyConf( Bunch ): + """ + Fake conf module for genetrack/atlas. + """ + pass + +if __name__ == "__main__": + options, args = doc_optparse.parse( __doc__ ) + try: + label = options.label + fit_name, fit_meta = options.fits.split(':')[0], [int(x)-1 for x in options.fits.split(':')[1:]] + fit_meta = Bunch(chromCol=fit_meta[0], positionCol=fit_meta[1], forwardCol=fit_meta[2], reverseCol=fit_meta[3]) + fit = ( label, fit_name, fit_meta, ) + # split apart the string into nested lists, preserves order + if options.feats: + feats = [ ( + feat_label, + fname, + Bunch(chromCol=int(chromCol)-1, startCol=int(startCol)-1, endCol=int(endCol)-1, + strandCol=int(strandCol)-1, nameCol=int(nameCol)-1), + ) + for feat_label, fname, chromCol, startCol, endCol, strandCol, nameCol + in ( feat.split(':') for feat in options.feats.split(',') )] + else: + feats = [] + data_dir = options.data + output = options.output + except: + doc_optparse.exception() + + main(label, fit, feats, data_dir, output) + \ No newline at end of file diff --git a/tools/visualization/genetrack.xml b/tools/visualization/genetrack.xml new file mode 100644 index 00000000000..65c5f519b31 --- /dev/null +++ b/tools/visualization/genetrack.xml @@ -0,0 +1,53 @@ + + + Track creator/viewer + + + + + + + genetrack.py -l $data_label + -1 ${fit_data}:${fit_data.metadata.chromCol}:${fit_data.metadata.positionCol}:${fit_data.metadata.forwardCol}:${fit_data.metadata.reverseCol} + #if $feature_data + -2 + #end if + #for $data in $feature_data + ${data.name}:${data.input}:${data.input.metadata.chromCol}:${data.input.metadata.startCol}:${data.input.metadata.endCol}:${data.input.metadata.strandCol}:${data.input.metadata.nameCol}, + #end for + -d ${genetrack.files_path} + -o ${bed_out} + + + + + [a-zA-Z0-9]{0,25} + + + + + + [a-zA-Z0-9]{0,25} + + + + + + + + + + +This tool takes the input Fit Data and creates a peak and curve plot showing +the reads and fitness on each basepair. Features can be plotted below as tracks. + +----- + +**Syntax** + +- **Track Label** is the name of the generated track. +- **Fit Data** are the datasets to calculate coverage/reads across basepairs and generate a curve. +- **Features** are additional datasets (interval format) to be plotted below as tracks. + + + diff --git a/tools/visualization/genetrack_code.py b/tools/visualization/genetrack_code.py new file mode 100644 index 00000000000..9c20ec27b73 --- /dev/null +++ b/tools/visualization/genetrack_code.py @@ -0,0 +1,13 @@ +import sets, os +from galaxy import eggs +from galaxy import jobs +from galaxy.tools.parameters import DataToolParameter + +def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None): + """ + Copy data_label to genetrack.metadata.label + """ + out_data['genetrack'].metadata.label = param_dict['data_label'] + out_data['genetrack'].info = "Use the link below to view the custom track." + out_data['bed_out'].info = "" + \ No newline at end of file