From 82d09f5824c43256114adebbc3f661ab8e3e9dce Mon Sep 17 00:00:00 2001 From: Ian Schenck Date: Wed, 21 Jan 2009 16:48:54 -0500 Subject: [PATCH] Finally remerged changeset. Should be good now. --- datatypes_conf.xml.sample | 3 + .../converters/interval_to_coverage.py | 80 +++++ .../converters/interval_to_coverage.xml | 18 ++ lib/galaxy/datatypes/coverage.py | 30 ++ lib/galaxy/datatypes/registry.py | 6 +- lib/galaxy/datatypes/tracks.py | 30 ++ lib/galaxy/web/base/controller.py | 5 +- lib/galaxy/web/buildapp.py | 7 +- lib/galaxy/web/controllers/genetrack.py | 283 ++++++++++++++++++ static/genetrack/genetrack.css | 78 +++++ static/genetrack/genetrack.js | 79 +++++ templates/genetrack/base.html | 29 ++ templates/genetrack/index.html | 75 +++++ templates/genetrack/search.html | 55 ++++ tool_conf.xml.sample | 3 + tools/sr_mapping/lastz_wrapper.xml | 2 +- tools/visualization/genetrack.py | 189 ++++++++++++ tools/visualization/genetrack.xml | 62 ++++ tools/visualization/genetrack_code.py | 13 + 19 files changed, 1042 insertions(+), 5 deletions(-) create mode 100644 lib/galaxy/datatypes/converters/interval_to_coverage.py create mode 100644 lib/galaxy/datatypes/converters/interval_to_coverage.xml create mode 100644 lib/galaxy/datatypes/coverage.py create mode 100644 lib/galaxy/datatypes/tracks.py create mode 100644 lib/galaxy/web/controllers/genetrack.py create mode 100644 static/genetrack/genetrack.css create mode 100644 static/genetrack/genetrack.js create mode 100644 templates/genetrack/base.html create mode 100644 templates/genetrack/index.html create mode 100644 templates/genetrack/search.html create mode 100644 tools/visualization/genetrack.py create mode 100644 tools/visualization/genetrack.xml create mode 100644 tools/visualization/genetrack_code.py diff --git a/datatypes_conf.xml.sample b/datatypes_conf.xml.sample index 58a8929f7a4..c7e15a4bde7 100644 --- a/datatypes_conf.xml.sample +++ b/datatypes_conf.xml.sample @@ -5,8 +5,10 @@ + + @@ -17,6 +19,7 @@ + diff --git a/lib/galaxy/datatypes/converters/interval_to_coverage.py b/lib/galaxy/datatypes/converters/interval_to_coverage.py new file mode 100644 index 00000000000..9b4bfccfa10 --- /dev/null +++ b/lib/galaxy/datatypes/converters/interval_to_coverage.py @@ -0,0 +1,80 @@ +#!/usr/bin/env python +""" +Converter to generate 3 (or 4) column base-pair coverage from an interval file. + +usage: %prog bed_file out_file + -1, --cols1=N,N,N,N: Columns for chrom, start, end, strand in interval file + -2, --cols2=N,N,N,N: Columns for chrom, start, end, strand in coverage file +""" +import sys +from galaxy import eggs +import pkg_resources; pkg_resources.require( "bx-python" ) +from bx.intervals import io +from bx.cookbook import doc_optparse + +INTERVAL_METADATA = ('chromCol', + 'startCol', + 'endCol', + 'strandCol',) + +COVERAGE_METADATA = ('chromCol', + 'positionCol', + 'forwardCol', + 'reverseCol',) + +def main( interval, coverage ): + chroms = dict() + for record in interval: + if not type( record ) is io.GenomicInterval: continue + chrom = chroms[record.chrom] = chroms.get(record.chrom, dict()) + for position in xrange(record.start, record.end): + coverages = chrom[position] = chrom.get(position,[0,0]) + if record.strand == "-": coverages[1] += 1 + else: coverages[0] += 1 + for chrom in sorted(chroms.iterkeys()): + positions = chroms[chrom] + for position in sorted(positions.iterkeys()): + coverage.write( chrom=chrom, position=position, forward=positions[position][0], reverse=positions[position][1] ) + +class CoverageWriter( object ): + def __init__( self, out_stream=None, chromCol=0, positionCol=1, forwardCol=2, reverseCol=3 ): + self.chromCol, self.positionCol, self.forwardCol, self.reverseCol = chromCol, positionCol, forwardCol, reverseCol + self.nfields = max( chromCol, positionCol, forwardCol, reverseCol )+1 + self.out_stream = out_stream + self.nlines = 0 + + def write(self, chrom="chr", position=0, forward=0, reverse=0 ): + self.nlines += 1 + if self.nlines % 64000: self.out_stream.flush() + outlist = [None] * self.nfields + outlist[self.chromCol] = str(chrom) + outlist[self.positionCol] = str(position) + if self.reverseCol == -1: outlist[self.forwardCol] = str(forward + reverse) + else: + outlist[self.forwardCol] = str(forward) + outlist[self.reverseCol] = str(reverse) + self.out_stream.write("%s\n" % "\t".join( outlist )) + + def flush(self): + self.out_stream.flush() + +if __name__ == "__main__": + options, args = doc_optparse.parse( __doc__ ) + try: + chr_col_1, start_col_1, end_col_1, strand_col_1 = [int(x)-1 for x in options.cols1.split(',')] + chr_col_2, position_col_2, forward_col_2, reverse_col_2 = [int(x)-1 for x in options.cols2.split(',')] + in_fname, out_fname = args + except: + doc_optparse.exception() + + coverage = CoverageWriter( out_stream = open(out_fname, "a"), + chromCol = chr_col_2, positionCol = position_col_2, + forwardCol = forward_col_2, reverseCol = reverse_col_2, ) + interval = io.NiceReaderWrapper( open(in_fname, "r"), + chrom_col=chr_col_1, + start_col=start_col_1, + end_col=end_col_1, + strand_col=strand_col_1, + fix_strand=True ) + main( interval, coverage ) + coverage.flush() \ No newline at end of file diff --git a/lib/galaxy/datatypes/converters/interval_to_coverage.xml b/lib/galaxy/datatypes/converters/interval_to_coverage.xml new file mode 100644 index 00000000000..3ad64f7b9ba --- /dev/null +++ b/lib/galaxy/datatypes/converters/interval_to_coverage.xml @@ -0,0 +1,18 @@ + + + + interval_to_coverage.py $input1 $output1 + -1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol} + -2 ${output1.metadata.chromCol},${output1.metadata.positionCol},${output1.metadata.forwardCol},${output1.metadata.reverseCol} + + + + + + + + + + + + diff --git a/lib/galaxy/datatypes/coverage.py b/lib/galaxy/datatypes/coverage.py new file mode 100644 index 00000000000..4bd76425a71 --- /dev/null +++ b/lib/galaxy/datatypes/coverage.py @@ -0,0 +1,30 @@ +""" +Coverage datatypes + +""" +import pkg_resources +pkg_resources.require( "bx-python" ) + +import logging, os, sys, time, sets, tempfile, shutil +import data +from galaxy import util +from galaxy.datatypes.sniff import * +from galaxy.web import url_for +from cgi import escape +import urllib +from bx.intervals.io import * +from galaxy.datatypes import metadata +from galaxy.datatypes.metadata import MetadataElement +from galaxy.datatypes.tabular import Tabular + +log = logging.getLogger(__name__) + +class LastzCoverage( Tabular ): + file_ext = "coverage" + + MetadataElement( name="chromCol", default=1, desc="Chrom column", param=metadata.ColumnParameter ) + MetadataElement( name="positionCol", default=2, desc="Position column", param=metadata.ColumnParameter ) + MetadataElement( name="forwardCol", default=3, desc="Forward or aggregate read column", param=metadata.ColumnParameter ) + MetadataElement( name="reverseCol", desc="Optional reverse read column", param=metadata.ColumnParameter, optional=True, no_value=0 ) + MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True, visible=False ) + \ No newline at end of file diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index b7cd9b58b7c..816851d7437 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -3,7 +3,7 @@ Provides mapping between extensions and datatypes, mime-types, etc. """ import os import logging -import data, tabular, interval, images, sequence, qualityscore, genetics, xml +import data, tabular, interval, images, sequence, qualityscore, genetics, xml, coverage, tracks import galaxy.util from galaxy.util.odict import odict @@ -94,12 +94,14 @@ class Registry( object ): 'bed' : interval.Bed(), 'binseq.zip' : images.Binseq(), 'blastxml' : xml.BlastXml(), + 'coverage' : coverage.LastzCoverage(), 'customtrack' : interval.CustomTrack(), 'csfasta' : sequence.csFasta(), 'fasta' : sequence.Fasta(), 'fastqsolexa' : sequence.FastqSolexa(), 'gff' : interval.Gff(), - 'gff3' : interval.Gff3(), + 'gff3' : interval.Gff3(), + 'genetrack' : tracks.GeneTrack(), 'interval' : interval.Interval(), 'laj' : images.Laj(), 'lav' : sequence.Lav(), diff --git a/lib/galaxy/datatypes/tracks.py b/lib/galaxy/datatypes/tracks.py new file mode 100644 index 00000000000..1c5a9291bef --- /dev/null +++ b/lib/galaxy/datatypes/tracks.py @@ -0,0 +1,30 @@ +""" +Datatype classes for tracks/track views within galaxy. +""" + +import data +import logging +import re +from cgi import escape +from galaxy.datatypes.metadata import MetadataElement +from galaxy.datatypes import metadata +import galaxy.model +from galaxy import util +from galaxy.web import url_for +from sniff import * + +log = logging.getLogger(__name__) + +class GeneTrack( data.Binary ): + file_ext = "genetrack" + + MetadataElement( name="hdf", default="data.hdf", desc="HDF DB", readonly=True, visible=True, no_value=0 ) + MetadataElement( name="sqlite", default="features.sqlite", desc="SQLite Features DB", readonly=True, visible=True, no_value=0 ) + MetadataElement( name="label", default="Custom", desc="Track Label", readonly=True, visible=True, no_value="Custom" ) + + def __init__(self, **kwargs): + super(GeneTrack, self).__init__(**kwargs) + self.add_display_app( 'genetrack', 'View in ', '', 'genetrack_link' ) + + def genetrack_link( self, dataset, type, app, base_url ): + return [('GeneTrack', url_for(controller='genetrack', action='index', dataset_id=dataset.id ))] \ No newline at end of file diff --git a/lib/galaxy/web/base/controller.py b/lib/galaxy/web/base/controller.py index 534124e2131..f2e95dfbcbd 100644 --- a/lib/galaxy/web/base/controller.py +++ b/lib/galaxy/web/base/controller.py @@ -28,4 +28,7 @@ class BaseController( object ): Root = BaseController """ Deprecated: `BaseController` used to be available under the name `Root` -""" \ No newline at end of file +""" + +class ControllerUnavailable( Exception ): + pass \ No newline at end of file diff --git a/lib/galaxy/web/buildapp.py b/lib/galaxy/web/buildapp.py index 797a4aca263..65a5d315ea8 100644 --- a/lib/galaxy/web/buildapp.py +++ b/lib/galaxy/web/buildapp.py @@ -28,13 +28,18 @@ def add_controllers( webapp, app ): them to the webapp. """ from galaxy.web.base.controller import BaseController + from galaxy.web.base.controller import ControllerUnavailable import galaxy.web.controllers controller_dir = galaxy.web.controllers.__path__[0] for fname in os.listdir( controller_dir ): if not( fname.startswith( "_" ) ) and fname.endswith( ".py" ): name = fname[:-3] module_name = "galaxy.web.controllers." + name - module = __import__( module_name ) + try: + module = __import__( module_name ) + except ControllerUnavailable, exc: + log.debug("%s could not be loaded: %s" % (module_name, str(exc))) + continue for comp in module_name.split( "." )[1:]: module = getattr( module, comp ) # Look for a controller inside the modules diff --git a/lib/galaxy/web/controllers/genetrack.py b/lib/galaxy/web/controllers/genetrack.py new file mode 100644 index 00000000000..f4b13e2b09c --- /dev/null +++ b/lib/galaxy/web/controllers/genetrack.py @@ -0,0 +1,283 @@ +import time, glob, os +from itertools import cycle + +from mako import exceptions +from mako.template import Template +from mako.lookup import TemplateLookup +from galaxy.web.base.controller import * + +try: + import pkg_resources + pkg_resources.require("GeneTrack") + import atlas + from atlas import sql + from atlas import hdf + from atlas import util as atlas_utils + from atlas.web import formlib, feature_query, feature_filter + from atlas.web import label_cache as atlas_label_cache + from atlas.plotting.const import * + from atlas.plotting.tracks import prefab + from atlas.plotting.tracks import chart + from atlas.plotting import tracks +except Exception, exc: + raise ControllerUnavailable("GeneTrack could not import a required dependency: %s" % str(exc)) + +pkg_resources.require( "Paste" ) +import paste.httpexceptions + +# Database helpers +SHOW_LABEL_LIMIT = 10000 +color = cycle( [LIGHT, WHITE] ) + +def list_labels(session): + """ + Returns a list of labels that will be plotted in order. + """ + labels = sql.Label + query = session.query(labels).order_by("-id") + return query + +def open_databases( conf ): + """ + A helper function that returns handles to the hdf and sql databases + """ + db = hdf.hdf_open( conf.HDF_DATABASE, mode='r' ) + session = sql.get_session( conf.SQL_URI ) + return db, session + +def hdf_query(db, name, param, autosize=False ): + """ + Schema specific hdf query. + Note that returns data as columns not rows. + """ + if not hdf.has_node(db=db, name=name): + atlas.warn( 'missing label %s' % name ) + return [], [], [], [] + data = hdf.GroupData( db=db, name=name) + istart, iend = data.get_indices(label=param.chrom, start=param.start, stop=param.end) + table = data.get_table(label=param.chrom) + if autosize: + # attempts to reduce the number of points + size = len( table.cols.ix[istart:iend] ) + step = max( [1, size/1200] ) + else: + step = 1 + + ix = table.cols.ix[istart:iend:step].tolist() + wx = table.cols.wx[istart:iend:step].tolist() + cx = table.cols.cx[istart:iend:step].tolist() + ax = table.cols.ax[istart:iend:step].tolist() + return ix, wx, cx, ax + +# Chart helpers +def build_tracks( param, conf, data_label, fit_label, pred_label, strand, show=False ): + """ + Builds tracks + """ + # gets all the labels for a fast lookup + label_cache = atlas_label_cache( conf ) + + # get database handles for hdf and sql + db, session = open_databases( conf ) + + # fetching x and y coordinates for bar and fit (line) for + # each strand plus (p), minus (m), all (a) + bix, bpy, bmy, bay = hdf_query( db=db, name=data_label, param=param ) + fix, fpy, fmy, fay = hdf_query( db=db, name=fit_label, param=param ) + + # close the hdf database + db.close() + + # get all features within the range + all = feature_query( session=session, param=param ) + + # draws the barchart and the nucleosome chart below it + if strand == 'composite': + bar = prefab.composite_bartrack( fix=fix, fay=fay, bix=bix, bay=bay, param=param) + else: + bar = prefab.twostrand_bartrack( fix=fix, fmy=fmy, fpy=fpy, bix=bix, bmy=bmy, bpy=bpy, param=param) + + charts = list() + charts.append( bar ) + + return charts + +def feature_chart(param=None, session=None, label=None, label_dict={}): + # draw the ORF tracks + all = feature_filter(feature_query(session=session, param=param), name=label, kdict=label_dict) + if len(all) == 0: return [] + opts = track_options( + xscale=param.xscale, w=param.width, fgColor=PURPLE, + show_labels=param.show_labels, ylabel=str(label), + bgColor=color.next() + ) + return [ + tracks.split_tracks(features=all, options=opts, split=param.show_labels, track_type='vector') + ] + +def consolidate_charts( charts, param ): + # create the multiplot + opt = chart_options( w=param.width ) + multi = chart.MultiChart(options=opt, charts=charts) + return multi + +# SETUP Track Builders +import functools +def twostrand_tracks( param=None, conf=None ): + return build_tracks( data_label=conf.LABEL, fit_label=conf.FIT_LABEL, pred_label=conf.PRED_LABEL, param=param, conf=conf, strand='twostrand') +def composite_tracks( param=None, conf=None ): + return build_tracks( data_label=conf.LABEL, fit_label=conf.FIT_LABEL, pred_label=conf.PRED_LABEL, param=param, conf=conf, strand='composite') + +class BaseConf( object ): + """ + Fake web_conf for atlas. + """ + IMAGE_DIR = "static/genetrack/plots/" + LEVELS = [str(x) for x in [ 50, 100, 250, 500, 1000, 2500, 5000, 10000, 20000, 50000, 100000, 200000 ]] + ZOOM_LEVELS = zip(LEVELS, LEVELS) + PLOT_SETUP = [ + ('comp-id', 'Composite' , 'genetrack/index.html', composite_tracks ), + ('two-id' , 'Two Strand', 'genetrack/index.html', twostrand_tracks ), + ] + PLOT_CHOICES = [ (id, name) for (id, name, page, func) in PLOT_SETUP ] + PLOT_MAPPER = dict( [ (id, (page, func)) for (id, name, page, func) in PLOT_SETUP ] ) + + def __init__(self, **kwds): + for key,value in kwds.items(): + setattr( self, key, value) + +class WebRoot(BaseController): + @web.expose + def search(self, trans, word='', dataset_id=None, submit=''): + """ + Default search page + """ + data = trans.app.model.HistoryDatasetAssociation.get( dataset_id ) + if not data: + raise paste.httpexceptions.HTTPRequestRangeNotSatisfiable( "Invalid reference dataset id: %s." % str( dataset_id ) ) + # the main configuration file + conf = BaseConf( + TITLE = "%s: %s" % (data.metadata.dbkey, data.metadata.label), + HDF_DATABASE = os.path.join( data.extra_files_path, data.metadata.hdf ), + SQL_URI = "sqlite:///%s" % os.path.join( data.extra_files_path, data.metadata.sqlite ), + LABEL = data.metadata.label, + FIT_LABEL = "%s-SIGMA-%d" % (data.metadata.label, 20), + PRED_LABEL = "PRED-%s-SIGMA-%d" % (data.metadata.label, 20), + ) + from atlas import hdf + db = hdf.hdf_open( conf.HDF_DATABASE, mode='r' ) + conf.CHROM_FIELDS = [(x,x) for x in hdf.GroupData(db=db, name=conf.LABEL).labels] + db.close() + + param = atlas.Param( word=word ) + # search with features based on param.feature + + # search for a given + session = sql.get_session( conf.SQL_URI ) + + if param.word: + def search_query( word, text ): + query = session.query(sql.Feature).filter( "name LIKE :word or freetext LIKE :text" ).params(word=word, text=text) + query = list(query[:20]) + return query + + # a little heuristics to match most likely target + targets = [ + (param.word+'%', 'No match'), # match beginning + ('%'+param.word+'%', 'No match'), # match name anywhere + ('%'+param.word+'%', '%'+param.word+'%'), # match json anywhere + ] + for word, text in targets: + query = search_query( word=word, text=text) + if query: + break + else: + query = [] + + return trans.fill_template_mako('genetrack/search.html', param=param, query=query, dataset_id=dataset_id) + + @web.expose + def index(self, trans, dataset_id=None, **kwds): + """ + Main request handler + """ + color = cycle( [LIGHT, WHITE] ) + data = trans.app.model.HistoryDatasetAssociation.get( dataset_id ) + if not data: + raise paste.httpexceptions.HTTPRequestRangeNotSatisfiable( "Invalid reference dataset id: %s." % str( dataset_id ) ) + # the main configuration file + conf = BaseConf( + TITLE = "%s: %s" % (data.metadata.dbkey, data.metadata.label), + HDF_DATABASE = os.path.join( data.extra_files_path, data.metadata.hdf ), + SQL_URI = "sqlite:///%s" % os.path.join( data.extra_files_path, data.metadata.sqlite ), + LABEL = data.metadata.label, + FIT_LABEL = "%s-SIGMA-%d" % (data.metadata.label, 20), + PRED_LABEL = "PRED-%s-SIGMA-%d" % (data.metadata.label, 20), + ) + session = sql.get_session( conf.SQL_URI ) + + if os.path.exists( conf.HDF_DATABASE ): + db = hdf.hdf_open( conf.HDF_DATABASE, mode='r' ) + conf.CHROM_FIELDS = [(x,x) for x in hdf.GroupData(db=db, name=conf.LABEL).labels] + db.close() + else: + query = session.execute(sql.select([sql.feature_table.c.chrom]).distinct()) + conf.CHROM_FIELDS = [(x.chrom,x.chrom) for x in query] + + # generate a new form based on the configuration + form = formlib.main_form( conf ) + + # clear the tempdir every once in a while + atlas_utils.clear_tempdir( dir=conf.IMAGE_DIR, days=1, chance=10) + + incoming = form.defaults() + incoming.update( kwds ) + + # manage the zoom and pan requests + incoming = formlib.zoom_change( kdict=incoming, levels=conf.LEVELS) + incoming = formlib.pan_view( kdict=incoming ) + + # process the form + param = atlas.Param( **incoming ) + form.process( incoming ) + + if kwds and form.isSuccessful(): + # adds the sucessfull parameters + param.update( form.values() ) + + # if it was a search word not a number go to search page + try: + center = int( param.feature ) + except ValueError: + # go and search for these + return trans.response.send_redirect( web.url_for( controller='genetrack', action='search', word=param.feature, dataset_id=dataset_id ) ) + + # keep image at a sane size + param.width = min( [2000, int(param.img_size)] ) + + # get the template and the function used to generate the tracks + tmpl_name, track_maker = conf.PLOT_MAPPER[param.plot] + + charts = [] + + fname, fpath = atlas_utils.make_tempfile( dir=conf.IMAGE_DIR, suffix='.png') + param.fname = fname + + # set the scale of the plot + param.xscale = [ param.start, param.end ] + + # when visualizing on wide scales labels are not useful + param.show_labels = ( param.end - param.start ) <= SHOW_LABEL_LIMIT + + if track_maker is not None and os.path.exists( conf.HDF_DATABASE ): + # generate the fit track + charts = track_maker( param=param, conf=conf ) + + for label in list_labels( session ): + charts.extend( feature_chart(param=param, session=session, label=label.name, label_dict={label.name:label.id}) ) + track_chart = consolidate_charts( charts, param ) + track_chart.save(fname=fpath) + + return trans.fill_template_mako(tmpl_name, conf=conf, form=form, param=param, dataset_id=dataset_id) + + diff --git a/static/genetrack/genetrack.css b/static/genetrack/genetrack.css new file mode 100644 index 00000000000..668300be06d --- /dev/null +++ b/static/genetrack/genetrack.css @@ -0,0 +1,78 @@ + +body { + font-family: "Trebuchet MS", Arial, tahoma, sans-serif; + font-size: 14px; + line-height: 1.6em; + margin: 0; + padding: 0; + border-top: 9px solid #CCD9FF; +} + +/* Error message style */ +.error{ + background: #FFFF66; +} + +/* Error message style */ +.message{ + background: #33FF66; +} + +/* Odd data row in the table */ +.selected { + background-color: #FFFFCC; +} + +.nav_button{ + background-color:#EEEEEE; + border:1px solid; + color: #000000; +} + +.nav_button:hover{ + background-color:#000000; + border:1px solid; + color: #FFFFFF; +} + +.grey { + background-color: #EFEFEF; +} + +.odd { + background-color: #ECECEC; +} + +.even { + background-color: #FFFFFF; +} + +/* Text table style */ +.data_table { + border: 1px solid #CCCCCC; + background-color: white; +} + +/* Footer is added to every page */ +#footer { + background: #EFEFEF; + text-align:center; + padding:.2em; + border-top: 1px solid #CCD9FF; + border-bottom: 1px solid #CCD9FF; + clear: both; +} + +#footer p { + font-size:.94em; line-height:2em; color:#cccccc; margin: 0; + } + +#tag { + font-size:.80em; margin: 4px; padding: 2px; + } + + +#footer img { + vertical-align: middle; margin-left: 3px; padding-bottom: 2px; +} + diff --git a/static/genetrack/genetrack.js b/static/genetrack/genetrack.js new file mode 100644 index 00000000000..be88d107ab8 --- /dev/null +++ b/static/genetrack/genetrack.js @@ -0,0 +1,79 @@ +var cookie_name = "genetrack_ui" +var now = new Date(); +now.setTime(now.getTime() + 365 * 24 * 60 * 60 * 1000); + +// this toggles between none and block +function toggle(name){ + var elem = get(name) + if (elem) { + if (elem.style.display=="none"){ + elem.style.display="block" + setCookie(cookie_name, name, now) + } else { + elem.style.display="none" + setCookie(cookie_name, '', now) + } + + } +} + +function main(){ + //executed upon main body load + var value = getCookie(cookie_name); + toggle( value ) +} + +// this toggles between visible and hidden +function show(name){ + var elem = get(name) + if (elem.style.visibility=="hidden"){ + elem.style.visibility="visible"; + } else { + elem.style.visibility="hidden"; + } +} + +// utility function to get the length of on object +function len(obj){ + return obj.length; +} + +// utility function to get an element by id +function get(name){ + return document.getElementById(name); +} + +// pops up a window +function pop_up(url) { + day = new Date(); + id = day.getTime(); + eval("page" + id + " = window.open(url, '" + id + "', 'toolbar=0,scrollbars=1,location=0,statusbar=1,menubar=0,resizable=1,width=500,height=300');"); +} + +// +// cookie management off the web +// http://www.webreference.com/js/column8/property.html +// +function setCookie(name, value, expires, path, domain, secure) { + var curCookie = name + "=" + escape(value) + + ((expires) ? "; expires=" + expires.toGMTString() : "") + + ((path) ? "; path=" + path : "") + + ((domain) ? "; domain=" + domain : "") + + ((secure) ? "; secure" : ""); + document.cookie = curCookie; +} + +function getCookie(name) { + var dc = document.cookie; + var prefix = name + "="; + var begin = dc.indexOf("; " + prefix); + if (begin == -1) { + begin = dc.indexOf(prefix); + if (begin != 0) return null; + } else + begin += 2; + var end = document.cookie.indexOf(";", begin); + if (end == -1) + end = dc.length; + return unescape(dc.substring(begin + prefix.length, end)); +} diff --git a/templates/genetrack/base.html b/templates/genetrack/base.html new file mode 100644 index 00000000000..cd381d54b93 --- /dev/null +++ b/templates/genetrack/base.html @@ -0,0 +1,29 @@ + + + + +${self.title()} + + + + +<%def name="title()"> + Title + + +<%def name="footer()"> + + + + + + ${self.body()} + ${self.footer()} + + + diff --git a/templates/genetrack/index.html b/templates/genetrack/index.html new file mode 100644 index 00000000000..23dee5cea67 --- /dev/null +++ b/templates/genetrack/index.html @@ -0,0 +1,75 @@ +## index.html +<%inherit file="base.html"/> +<%def name="title()"> + Index + + +

${conf.TITLE}

+ +
+ + + + + % if form.errors(): + + % endif + + + + + + + + + + + + + + + + + +
+ % for ekey, evalue in form.errors().items(): + ERROR:    ${ekey}:    ${evalue}
+ % endfor +
+ More + + Chromosome: ${form.chrom.tag()}   + Feature: ${form.feature.tag()}   + Width: ${form.zoom.tag()}   + Plot: ${form.plot.tag()}   + + + +
+ +     + +     + +     + +
+ +
+ +
+ + +
diff --git a/templates/genetrack/search.html b/templates/genetrack/search.html new file mode 100644 index 00000000000..d707a6da479 --- /dev/null +++ b/templates/genetrack/search.html @@ -0,0 +1,55 @@ +## search.html +<%! +from itertools import cycle +colors = cycle( [ 'even', 'odd' ] ) +%> + +<%inherit file="base.html"/> +<%def name="title()"> + Search + + +

Search

+ +
+
+ Search terms + + +
+
+ +% if param.word: + + % if len(query)>0: +

Showing the best ${len(query)} matches

+ + + + + % for color, row in zip(colors, query): + ${makerow(color, row)} + % endfor +
Name + Chromosome + Start:End + Type +
+ + % else: +

No results found

+ % endif + +%endif + +
+<%def name="makerow(color, row)"> + + ${row.name} + ${row.chrom} + ${row.start}:${row.end} + ${row.label.name} + + + + diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample index bb1dcbab138..b70330e3a0d 100644 --- a/tool_conf.xml.sample +++ b/tool_conf.xml.sample @@ -303,4 +303,7 @@ +
+ +
diff --git a/tools/sr_mapping/lastz_wrapper.xml b/tools/sr_mapping/lastz_wrapper.xml index e3aabf57243..f29d5fb7168 100644 --- a/tools/sr_mapping/lastz_wrapper.xml +++ b/tools/sr_mapping/lastz_wrapper.xml @@ -82,7 +82,7 @@ - + lastz diff --git a/tools/visualization/genetrack.py b/tools/visualization/genetrack.py new file mode 100644 index 00000000000..14c290f05d2 --- /dev/null +++ b/tools/visualization/genetrack.py @@ -0,0 +1,189 @@ +#!/usr/bin/env python +""" +Run GeneTrack(atlas) with a faked conf file to generate GeneTrack data files. + +usage: %prog + -l, --label=N: Data label for fit curve/peak plot + -1, --fits=N/N/N/N/N,...: Data files (interval format) for fit curve/peak plot + -2, --feats=N:M/N/N/N/N/N,...: Data files (interval format) for features. + -d, --data=N: Output path for hdf5 and sqlite databases. + -o, --output=N: Output path for export file. +""" +from galaxy import eggs +import pkg_resources +pkg_resources.require("GeneTrack") +pkg_resources.require("bx-python") + +import commands as oscommands +from atlas import commands +from atlas import sql +from bx.cookbook import doc_optparse +from bx.intervals import io + +import os +import tempfile +from functools import partial + +SIGMA = 20 +WIDTH = 5 * SIGMA +EXCLUSION_ZONE = 147 + +def main(label, fit, feats, data_dir, output): + os.mkdir(data_dir) + conf = DummyConf( + __name__=label, + CLOBBER = True, + DATA_SIZE = 3*10**6, + MINIMUM_PEAK_SIZE = 0.1, + LOADER_ENABLED = False, + FITTER_ENABLED = False, + PREDICTOR_ENABLED = False, + EXPORTER_ENABLED = False, + LOADER = loader, + FITTER = fitter, + PREDICTOR = predictor, + EXPORTER = partial( commands.exporter, formatter=commands.bed_formatter), + HDF_DATABASE = os.path.join( data_dir, "data.hdf" ), + SQL_URI = "sqlite:///%s" % os.path.join( data_dir, "features.sqlite" ), + SIGMA = SIGMA, + WIDTH = WIDTH, + DATA_LABEL = label, + FIT_LABEL = "%s-SIGMA-%d" % ( label,SIGMA ), + PEAK_LABEL = "PRED-%s-SIGMA-%d" % ( label,SIGMA ), + EXCLUSION_ZONE = EXCLUSION_ZONE, + LEFT_SHIFT = EXCLUSION_ZONE / 2, + RIGHT_SHIFT = EXCLUSION_ZONE / 2, + EXPORT_LABELS = [ "PRED-%s-SIGMA-%d" % ( label,SIGMA ) ], + EXPORT_DIR = os.path.join( data_dir ), + DATA_FILE=fit and fit[1] or None, + fit=fit, + feats=feats, + ) + if fit: + # Turn on fit processing. + conf.LOADER_ENABLED = True, + conf.FITTER_ENABLED = True, + conf.PREDICTOR_ENABLED = True, + conf.EXPORTER_ENABLED = True, + for feat in feats: + load_feature_files(conf, feats) + commands.execute(conf) + outname = "%s.%s.txt" % (conf.__name__, conf.EXPORT_LABELS[0] ) + if os.path.exists( os.path.join(data_dir, outname) ): + os.rename( os.path.join(data_dir, outname), output) + +# mod454 seems to be a module without a package. The necessary funcitons are +# stubbed out here until I'm sure of their final home. INS + +def loader( conf ): + from atlas import hdf + from mod454.schema import Mod454Schema as Schema + last_chrom = table = None + db = hdf.hdf_open( conf.HDF_DATABASE, mode='a', title='HDF database') + gp = hdf.create_group( db=db, name=conf.DATA_LABEL, desc='data group', clobber=conf.CLOBBER ) + fit_meta = conf.fit[2] + # iterate over the file and insert into table + for line in open( conf.fit[1], "r" ): + if line.startswith("chrom"): continue #Skip possible header + if line.startswith("#"): continue + fields = line.rstrip('\r\n').split('\t') + chrom = fields[fit_meta.chromCol] + if chrom != last_chrom: + if table: table.flush() + table = hdf.create_table( db=db, name=chrom, where=gp, schema=Schema, clobber=False ) + last_chrom = chrom + try: + position = int(fields[fit_meta.positionCol]) + forward = float(fields[fit_meta.forwardCol]) + reverse = fit_meta.reverseCol > -1 and float(fields[fit_meta.reverseCol]) or 0.0 + row = ( position, forward, reverse, forward+reverse, ) + table.append( [ row ] ) + except ValueError: + # Ignore bad lines + pass + table.flush() + db.close() + +def fitter( conf ): + from mod454.fitter import fitter as mod454_fitter + return mod454_fitter( conf ) + +def predictor( conf ): + from mod454.predictor import predictor as mod454_predictor + return mod454_predictor( conf ) + +def load_feature_files( conf, feats): + """ + Loads features from file names + """ + engine = sql.get_engine( conf.SQL_URI ) + sql.drop_indices(engine) + conn = engine.connect() + for label, fname, col_spec in feats: + label_id = sql.make_label(engine, name=label, clobber=False) + reader = io.NiceReaderWrapper( open(fname,"r"), + chrom_col=col_spec.chromCol, + start_col=col_spec.startCol, + end_col=col_spec.endCol, + strand_col=col_spec.strandCol, + fix_strand=False ) + values = list() + for interval in reader: + print interval + if not type( interval ) is io.GenomicInterval: continue + row = {'label_id':label_id, + 'name':col_spec.nameCol == -1 and "%s-%s" % (str(interval.start), str(interval.end)) or interval.fields[col_spec.nameCol], + 'altname':"", + 'chrom':interval.chrom, + 'start':interval.start, + 'end':interval.end, + 'strand':interval.strand, + 'value':0, + 'freetext':""} + values.append(row) + insert = sql.feature_table.insert() + conn.execute( insert, values) + conn.close() + sql.create_indices(engine) + + +class Bunch( object ): + def __init__(self, **kwargs): + for key,value in kwargs.items(): + setattr( self, key, value ) + +class DummyConf( Bunch ): + """ + Fake conf module for genetrack/atlas. + """ + pass + +if __name__ == "__main__": + options, args = doc_optparse.parse( __doc__ ) + try: + label = options.label + if options.fits: + fit_name, fit_meta = options.fits.split(':')[0], [int(x)-1 for x in options.fits.split(':')[1:]] + fit_meta = Bunch(chromCol=fit_meta[0], positionCol=fit_meta[1], forwardCol=fit_meta[2], reverseCol=fit_meta[3]) + fit = ( label, fit_name, fit_meta, ) + else: + fit = [] + # split apart the string into nested lists, preserves order + if options.feats: + feats = [ ( + feat_label, + fname, + Bunch(chromCol=int(chromCol)-1, startCol=int(startCol)-1, endCol=int(endCol)-1, + strandCol=int(strandCol)-1, nameCol=int(nameCol)-1), + ) + for feat_label, fname, chromCol, startCol, endCol, strandCol, nameCol + in ( feat.split(':') for feat in options.feats.split(',') if len(feat) > 0 )] + else: + feats = [] + data_dir = options.data + output = options.output + except: + doc_optparse.exception() + + main(label, fit, feats, data_dir, output) + \ No newline at end of file diff --git a/tools/visualization/genetrack.xml b/tools/visualization/genetrack.xml new file mode 100644 index 00000000000..8ab46560391 --- /dev/null +++ b/tools/visualization/genetrack.xml @@ -0,0 +1,62 @@ + + + Track creator/viewer + + + + + + + genetrack.py -l $data_label + #if not str($fit_data) == "None" + -1 + ${fit_data}:${fit_data.metadata.chromCol}:${fit_data.metadata.positionCol}:${fit_data.metadata.forwardCol}:${fit_data.metadata.reverseCol} + #end if + #if $feature_data + -2 + #end if + #for $data in $feature_data + ${data.name}:${data.input}:${data.input.metadata.chromCol}:${data.input.metadata.startCol}:${data.input.metadata.endCol}:${data.input.metadata.strandCol}:${data.input.metadata.nameCol}, + #end for + -d ${genetrack.files_path} + -o ${bed_out} + + + + + [a-zA-Z0-9]{0,25} + + + + + + [a-zA-Z0-9]{0,25} + + + + + + + + + + + tables + atlas + pychartdir + numpy + + +This tool takes the input Fit Data and creates a peak and curve plot showing +the reads and fitness on each basepair. Features can be plotted below as tracks. + +----- + +**Syntax** + +- **Track Label** is the name of the generated track. +- **Fit Data** are the datasets to calculate coverage/reads across basepairs and generate a curve. +- **Features** are additional datasets (interval format) to be plotted below as tracks. + + + diff --git a/tools/visualization/genetrack_code.py b/tools/visualization/genetrack_code.py new file mode 100644 index 00000000000..9c20ec27b73 --- /dev/null +++ b/tools/visualization/genetrack_code.py @@ -0,0 +1,13 @@ +import sets, os +from galaxy import eggs +from galaxy import jobs +from galaxy.tools.parameters import DataToolParameter + +def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None): + """ + Copy data_label to genetrack.metadata.label + """ + out_data['genetrack'].metadata.label = param_dict['data_label'] + out_data['genetrack'].info = "Use the link below to view the custom track." + out_data['bed_out'].info = "" + \ No newline at end of file