Initial GeneTrack commit. Most parts are in, but dependencies will still be a problem.

This commit is contained in:
Ian Schenck
2009-01-19 17:45:28 -05:00
parent 7c57ddc959
commit 6a0aa166aa
13 changed files with 594 additions and 5 deletions
+2
View File
@@ -82,6 +82,8 @@
<datatype extension="garnier" type="galaxy.datatypes.data:Text"/>
<datatype extension="gcg" type="galaxy.datatypes.data:Text"/>
<datatype extension="geecee" type="galaxy.datatypes.data:Text"/>
<datatype extension="genetrack" type="galaxy.datatypes.tracks:GeneTrack"/>
<datatype extension="coverage" type="galaxy.datatypes.coverage:LastzCoverage"/>
<datatype extension="genbank" type="galaxy.datatypes.data:Text"/>
<datatype extension="helixturnhelix" type="galaxy.datatypes.data:Text"/>
<datatype extension="hennig86" type="galaxy.datatypes.data:Text"/>
+30
View File
@@ -0,0 +1,30 @@
"""
Coverage datatypes
"""
import pkg_resources
pkg_resources.require( "bx-python" )
import logging, os, sys, time, sets, tempfile, shutil
import data
from galaxy import util
from galaxy.datatypes.sniff import *
from galaxy.web import url_for
from cgi import escape
import urllib
from bx.intervals.io import *
from galaxy.datatypes import metadata
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.tabular import Tabular
log = logging.getLogger(__name__)
class LastzCoverage( Tabular ):
file_ext = "coverage"
MetadataElement( name="chromCol", default=1, desc="Chrom column", param=metadata.ColumnParameter )
MetadataElement( name="positionCol", default=2, desc="Position column", param=metadata.ColumnParameter )
MetadataElement( name="forwardCol", default=3, desc="Forward or aggregate read column", param=metadata.ColumnParameter )
MetadataElement( name="reverseCol", desc="Optional reverse read column", param=metadata.ColumnParameter, optional=True, no_value=0 )
MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True, visible=False )
+4 -2
View File
@@ -3,7 +3,7 @@ Provides mapping between extensions and datatypes, mime-types, etc.
"""
import os
import logging
import data, tabular, interval, images, sequence, qualityscore, genetics, xml
import data, tabular, interval, images, sequence, qualityscore, genetics, xml, coverage, tracks
import galaxy.util
from galaxy.util.odict import odict
@@ -97,12 +97,14 @@ class Registry( object ):
'bed' : interval.Bed(),
'binseq.zip' : images.Binseq(),
'blastxml' : xml.BlastXml(),
'coverage' : coverage.LastzCoverage(),
'customtrack' : interval.CustomTrack(),
'csfasta' : sequence.csFasta(),
'fasta' : sequence.Fasta(),
'fastqsolexa' : sequence.FastqSolexa(),
'gff' : interval.Gff(),
'gff3' : interval.Gff3(),
'gff3' : interval.Gff3(),
'genetrack' : tracks.GeneTrack(),
'interval' : interval.Interval(),
'laj' : images.Laj(),
'lav' : sequence.Lav(),
+30
View File
@@ -0,0 +1,30 @@
"""
Datatype classes for tracks/track views within galaxy.
"""
import data
import logging
import re
from cgi import escape
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes import metadata
import galaxy.model
from galaxy import util
from galaxy.web import url_for
from sniff import *
log = logging.getLogger(__name__)
class GeneTrack( data.Binary ):
file_ext = "genetrack"
MetadataElement( name="hdf", default="data.hdf", desc="HDF DB", readonly=True, visible=True, no_value=0 )
MetadataElement( name="sqlite", default="features.sqlite", desc="SQLite Features DB", readonly=True, visible=True, no_value=0 )
MetadataElement( name="label", default="Custom", desc="Track Label", readonly=True, visible=True, no_value="Custom" )
def __init__(self, **kwargs):
super(GeneTrack, self).__init__(**kwargs)
self.add_display_app( 'genetrack', 'View in ', '', 'genetrack_link' )
def genetrack_link( self, dataset, type, app, base_url ):
return [('GeneTrack', url_for(controller='genetrack', action='index', dataset_id=dataset.id ))]
+161
View File
@@ -0,0 +1,161 @@
import time, glob, os
import pkg_resources
pkg_resources.require("GeneTrack")
import atlas
from atlas import sql
from atlas import util as atlas_utils
from atlas.web import formlib
from mako import exceptions
from mako.template import Template
from mako.lookup import TemplateLookup
from galaxy.web.base.controller import *
pkg_resources.require( "Paste" )
import paste.httpexceptions
# SETUP Track Builders
from mod454.trackbuilder import build_tracks
import functools
def twostrand_tracks( param=None, conf=None ):
return build_tracks( data_label=conf.LABEL, fit_label=conf.FIT_LABEL, pred_label=conf.PRED_LABEL, param=param, conf=conf, strand='twostrand')
def composite_tracks( param=None, conf=None ):
return build_tracks( data_label=conf.LABEL, fit_label=conf.FIT_LABEL, pred_label=conf.PRED_LABEL, param=param, conf=conf, strand='composite')
class BaseConf( object ):
"""
Fake web_conf for atlas.
"""
IMAGE_DIR = "static/genetrack/plots/"
LEVELS = [str(x) for x in [ 50, 100, 250, 500, 1000, 2500, 5000, 10000, 20000, 50000, 100000, 200000 ]]
ZOOM_LEVELS = zip(LEVELS, LEVELS)
PLOT_SETUP = [
('comp-id', 'Composite' , 'genetrack/index.html', composite_tracks ),
('two-id' , 'Two Strand', 'genetrack/index.html', twostrand_tracks ),
]
PLOT_CHOICES = [ (id, name) for (id, name, page, func) in PLOT_SETUP ]
PLOT_MAPPER = dict( [ (id, (page, func)) for (id, name, page, func) in PLOT_SETUP ] )
def __init__(self, **kwds):
for key,value in kwds.items():
setattr( self, key, value)
class WebRoot(BaseController):
@web.expose
def search(self, trans, word='', dataset_id=None, submit=''):
"""
Default search page
"""
data = trans.app.model.HistoryDatasetAssociation.get( dataset_id )
if not data:
raise paste.httpexceptions.HTTPRequestRangeNotSatisfiable( "Invalid reference dataset id: %s." % str( dataset_id ) )
# the main configuration file
conf = BaseConf(
TITLE = "<i>%s</i>: %s" % (data.metadata.dbkey, data.metadata.label),
HDF_DATABASE = os.path.join( data.extra_files_path, data.metadata.hdf ),
SQL_URI = "sqlite:///%s" % os.path.join( data.extra_files_path, data.metadata.sqlite ),
LABEL = data.metadata.label,
FIT_LABEL = "%s-SIGMA-%d" % (data.metadata.label, 20),
PRED_LABEL = "PRED-%s-SIGMA-%d" % (data.metadata.label, 20),
)
from atlas import hdf
db = hdf.hdf_open( conf.HDF_DATABASE, mode='r' )
conf.CHROM_FIELDS = [(x,x) for x in hdf.GroupData(db=db, name=conf.LABEL).labels]
db.close()
param = atlas.Param( word=word )
# search with features based on param.feature
# search for a given
session = sql.get_session( conf.SQL_URI )
if param.word:
def search_query( word, text ):
query = session.query(sql.Feature).filter( "name LIKE :word or freetext LIKE :text" ).params(word=word, text=text)
query = list(query[:20])
return query
# a little heuristics to match most likely target
targets = [
(param.word+'%', 'No match'), # match beginning
('%'+param.word+'%', 'No match'), # match name anywhere
('%'+param.word+'%', '%'+param.word+'%'), # match json anywhere
]
for word, text in targets:
query = search_query( word=word, text=text)
if query:
break
else:
query = []
return trans.fill_template_mako('genetrack/search.html', param=param, query=query, dataset_id=dataset_id)
@web.expose
def index(self, trans, dataset_id=None, **kwds):
"""
Main request handler
"""
data = trans.app.model.HistoryDatasetAssociation.get( dataset_id )
if not data:
raise paste.httpexceptions.HTTPRequestRangeNotSatisfiable( "Invalid reference dataset id: %s." % str( dataset_id ) )
# the main configuration file
conf = BaseConf(
TITLE = "<i>%s</i>: %s" % (data.metadata.dbkey, data.metadata.label),
HDF_DATABASE = os.path.join( data.extra_files_path, data.metadata.hdf ),
SQL_URI = "sqlite:///%s" % os.path.join( data.extra_files_path, data.metadata.sqlite ),
LABEL = data.metadata.label,
FIT_LABEL = "%s-SIGMA-%d" % (data.metadata.label, 20),
PRED_LABEL = "PRED-%s-SIGMA-%d" % (data.metadata.label, 20),
)
from atlas import hdf
db = hdf.hdf_open( conf.HDF_DATABASE, mode='r' )
conf.CHROM_FIELDS = [(x,x) for x in hdf.GroupData(db=db, name=conf.LABEL).labels]
db.close()
# generate a new form based on the configuration
form = formlib.main_form( conf )
# clear the tempdir every once in a while
atlas_utils.clear_tempdir( dir=conf.IMAGE_DIR, days=1, chance=10)
incoming = form.defaults()
incoming.update( kwds )
# manage the zoom and pan requests
incoming = formlib.zoom_change( kdict=incoming, levels=conf.LEVELS)
incoming = formlib.pan_view( kdict=incoming )
# process the form
param = atlas.Param( **incoming )
form.process( incoming )
if kwds and form.isSuccessful():
# adds the sucessfull parameters
param.update( form.values() )
# if it was a search word not a number go to search page
try:
center = int( param.feature )
except ValueError:
# go and search for these
return trans.response.send_redirect( web.url_for( controller='genetrack', action='search', word=param.feature, dataset_id=dataset_id ) )
# keep image at a sane size
param.width = min( [2000, int(param.img_size)] )
# get the template and the function used to generate the tracks
tmpl_name, track_maker = conf.PLOT_MAPPER[param.plot]
if track_maker is not None:
# generate the name that the image will be stored at
fname, fpath = atlas_utils.make_tempfile( dir=conf.IMAGE_DIR, suffix='.png')
param.fname = fname
# generate the track
track_chart = track_maker( param=param, conf=conf )
track_chart.save(fname=fpath)
return trans.fill_template_mako(tmpl_name, conf=conf, form=form, param=param, dataset_id=dataset_id)
+1 -1
View File
@@ -12,7 +12,7 @@ assert sys.version_info[:2] >= ( 2, 4 )
new_path = [ os.path.join( os.getcwd(), "lib" ) ]
new_path.extend( sys.path[1:] ) # remove scripts/ from the path
sys.path = new_path
print sys.path
from galaxy import eggs
import pkg_resources
+78
View File
@@ -0,0 +1,78 @@
body {
font-family: "Trebuchet MS", Arial, tahoma, sans-serif;
font-size: 14px;
line-height: 1.6em;
margin: 0;
padding: 0;
border-top: 9px solid #CCD9FF;
}
/* Error message style */
.error{
background: #FFFF66;
}
/* Error message style */
.message{
background: #33FF66;
}
/* Odd data row in the table */
.selected {
background-color: #FFFFCC;
}
.nav_button{
background-color:#EEEEEE;
border:1px solid;
color: #000000;
}
.nav_button:hover{
background-color:#000000;
border:1px solid;
color: #FFFFFF;
}
.grey {
background-color: #EFEFEF;
}
.odd {
background-color: #ECECEC;
}
.even {
background-color: #FFFFFF;
}
/* Text table style */
.data_table {
border: 1px solid #CCCCCC;
background-color: white;
}
/* Footer is added to every page */
#footer {
background: #EFEFEF;
text-align:center;
padding:.2em;
border-top: 1px solid #CCD9FF;
border-bottom: 1px solid #CCD9FF;
clear: both;
}
#footer p {
font-size:.94em; line-height:2em; color:#cccccc; margin: 0;
}
#tag {
font-size:.80em; margin: 4px; padding: 2px;
}
#footer img {
vertical-align: middle; margin-left: 3px; padding-bottom: 2px;
}
+79
View File
@@ -0,0 +1,79 @@
var cookie_name = "genetrack_ui"
var now = new Date();
now.setTime(now.getTime() + 365 * 24 * 60 * 60 * 1000);
// this toggles between none and block
function toggle(name){
var elem = get(name)
if (elem) {
if (elem.style.display=="none"){
elem.style.display="block"
setCookie(cookie_name, name, now)
} else {
elem.style.display="none"
setCookie(cookie_name, '', now)
}
}
}
function main(){
//executed upon main body load
var value = getCookie(cookie_name);
toggle( value )
}
// this toggles between visible and hidden
function show(name){
var elem = get(name)
if (elem.style.visibility=="hidden"){
elem.style.visibility="visible";
} else {
elem.style.visibility="hidden";
}
}
// utility function to get the length of on object
function len(obj){
return obj.length;
}
// utility function to get an element by id
function get(name){
return document.getElementById(name);
}
// pops up a window
function pop_up(url) {
day = new Date();
id = day.getTime();
eval("page" + id + " = window.open(url, '" + id + "', 'toolbar=0,scrollbars=1,location=0,statusbar=1,menubar=0,resizable=1,width=500,height=300');");
}
//
// cookie management off the web
// http://www.webreference.com/js/column8/property.html
//
function setCookie(name, value, expires, path, domain, secure) {
var curCookie = name + "=" + escape(value) +
((expires) ? "; expires=" + expires.toGMTString() : "") +
((path) ? "; path=" + path : "") +
((domain) ? "; domain=" + domain : "") +
((secure) ? "; secure" : "");
document.cookie = curCookie;
}
function getCookie(name) {
var dc = document.cookie;
var prefix = name + "=";
var begin = dc.indexOf("; " + prefix);
if (begin == -1) {
begin = dc.indexOf(prefix);
if (begin != 0) return null;
} else
begin += 2;
var end = document.cookie.indexOf(";", begin);
if (end == -1)
end = dc.length;
return unescape(dc.substring(begin + prefix.length, end));
}
+3 -1
View File
@@ -11,7 +11,6 @@
<tool file="data_source/wormbase.xml" />
<tool file="data_source/wormbase_test.xml" />
<tool file="data_source/flymine.xml" />
<tool file="data_source/flymine_test.xml" />
<tool file="data_source/encode_db.xml" />
<tool file="data_source/epigraph_import.xml" />
<tool file="data_source/epigraph_import_test.xml" />
@@ -302,4 +301,7 @@
<tool file="metag_tools/blat_wrapper.xml" />
<tool file="metag_tools/mapping_to_ucsc.xml" />
</section>
<section name="Tracks" id="tracks">
<tool file="visualization/genetrack.xml" />
</section>
</toolbox>
+1 -1
View File
@@ -82,7 +82,7 @@
<when input="out_format" value="maf" format="maf" />
</change_format>
</data>
<data format="tabular" name="output2" />
<data format="coverage" name="output2" />
</outputs>
<requirements>
<requirement type="binary">lastz</requirement>
+139
View File
@@ -0,0 +1,139 @@
#!/usr/bin/env python
"""
Run GeneTrack(atlas) with a faked conf file to generate GeneTrack data files.
usage: %prog
-l, --label=N: Data label for fit curve/peak plot
-1, --fits=N/N/N/N/N,...: Data files (interval format) for fit curve/peak plot
-2, --feats=N:M/N/N/N/N/N,...: Data files (interval format) for features.
-d, --data=N: Output path for hdf5 and sqlite databases.
-o, --output=N: Output path for export file.
"""
from galaxy import eggs
import pkg_resources
pkg_resources.require("GeneTrack")
pkg_resources.require("bx-python")
from atlas import commands
from bx.cookbook import doc_optparse
import os
import commands as oscommands
import tempfile
SIGMA = 20
WIDTH = 5 * SIGMA
EXCLUSION_ZONE = 147
def main(label, fit, feats, data_dir, output):
os.mkdir(data_dir)
conf = DummyConf(
__name__=label,
CLOBBER = True,
DATA_SIZE = 3*10**6,
MINIMUM_PEAK_SIZE = 0.1,
LOADER_ENABLED = True,
FITTER_ENABLED = True,
PREDICTOR_ENABLED = True,
EXPORTER_ENABLED = True,
LOADER = loader,
FITTER = fitter,
PREDICTOR = predictor,
EXPORTER = exporter,
HDF_DATABASE = os.path.join( data_dir, "data.hdf" ),
SQL_URI = "sqlite:///%s" % os.path.join( data_dir, "features.sqlite" ),
SIGMA = SIGMA,
WIDTH = WIDTH,
DATA_LABEL = label,
FIT_LABEL = "%s-SIGMA-%d" % ( label,SIGMA ),
PEAK_LABEL = "PRED-%s-SIGMA-%d" % ( label,SIGMA ),
EXCLUSION_ZONE = EXCLUSION_ZONE,
LEFT_SHIFT = EXCLUSION_ZONE / 2,
RIGHT_SHIFT = EXCLUSION_ZONE / 2,
EXPORT_LABELS = [ "PRED-%s-SIGMA-%d" % ( label,SIGMA ) ],
EXPORT_DIR = os.path.join( data_dir ),
DATA_FILE=fit[1],
fit=fit,
feats=feats,
)
commands.execute(conf)
# mod454 seems to be a module without a package. The necessary funcitons are
# stubbed out here until I'm sure of their final home. INS
def loader( conf ):
from atlas import hdf
from mod454.schema import Mod454Schema as Schema
last_chrom = table = None
db = hdf.hdf_open( conf.HDF_DATABASE, mode='a', title='HDF database')
gp = hdf.create_group( db=db, name=conf.DATA_LABEL, desc='data group', clobber=conf.CLOBBER )
fit_meta = conf.fit[2]
# iterate over the file and insert into table
for line in open( conf.fit[1], "r" ):
if line.startswith("chrom"): continue #Skip possible header
if line.startswith("#"): continue
fields = line.rstrip('\r\n').split('\t')
chrom = fields[fit_meta.chromCol]
if chrom != last_chrom:
if table: table.flush()
table = hdf.create_table( db=db, name=chrom, where=gp, schema=Schema, clobber=False )
last_chrom = chrom
try:
position = int(fields[fit_meta.positionCol])
forward = float(fields[fit_meta.forwardCol])
reverse = fit_meta.reverseCol > -1 and float(fields[fit_meta.reverseCol]) or 0.0
row = ( position, forward, reverse, forward+reverse, )
table.append( [ row ] )
except ValueError:
# Ignore bad lines
pass
table.flush()
db.close()
def fitter( conf ):
from mod454.fitter import fitter as mod454_fitter
return mod454_fitter( conf )
def predictor( conf ):
from mod454.predictor import predictor as mod454_predictor
return mod454_predictor( conf )
def exporter( conf ):
return commands.bed_exporter(conf)
class Bunch( object ):
def __init__(self, **kwargs):
for key,value in kwargs.items():
setattr( self, key, value )
class DummyConf( Bunch ):
"""
Fake conf module for genetrack/atlas.
"""
pass
if __name__ == "__main__":
options, args = doc_optparse.parse( __doc__ )
try:
label = options.label
fit_name, fit_meta = options.fits.split(':')[0], [int(x)-1 for x in options.fits.split(':')[1:]]
fit_meta = Bunch(chromCol=fit_meta[0], positionCol=fit_meta[1], forwardCol=fit_meta[2], reverseCol=fit_meta[3])
fit = ( label, fit_name, fit_meta, )
# split apart the string into nested lists, preserves order
if options.feats:
feats = [ (
feat_label,
fname,
Bunch(chromCol=int(chromCol)-1, startCol=int(startCol)-1, endCol=int(endCol)-1,
strandCol=int(strandCol)-1, nameCol=int(nameCol)-1),
)
for feat_label, fname, chromCol, startCol, endCol, strandCol, nameCol
in ( feat.split(':') for feat in options.feats.split(',') )]
else:
feats = []
data_dir = options.data
output = options.output
except:
doc_optparse.exception()
main(label, fit, feats, data_dir, output)
+53
View File
@@ -0,0 +1,53 @@
<tool id="genetrack1" name="GeneTrack">
<description>Track creator/viewer</description>
<code file="genetrack_code.py">
<hook exec_after_process="exec_after_process" />
</code>
<command interpreter="python">
genetrack.py -l $data_label
-1 ${fit_data}:${fit_data.metadata.chromCol}:${fit_data.metadata.positionCol}:${fit_data.metadata.forwardCol}:${fit_data.metadata.reverseCol}
#if $feature_data
-2
#end if
#for $data in $feature_data
${data.name}:${data.input}:${data.input.metadata.chromCol}:${data.input.metadata.startCol}:${data.input.metadata.endCol}:${data.input.metadata.strandCol}:${data.input.metadata.nameCol},
#end for
-d ${genetrack.files_path}
-o ${bed_out}
</command>
<inputs>
<param name="data_label" type="text" label="Track Label" size="50">
<validator type="regex" message="Please name the track with only alphanumeric characters.">[a-zA-Z0-9]{0,25}</validator>
</param>
<param name="fit_data" type="data" format="coverage" label="Coverage Dataset" />
<repeat name="feature_data" title="Features">
<param name="input" type="data" format="interval" label="Dataset" />
<param name="name" type="text" label="Feature Type (mRNA, ESTs, ORFs, etc.)" size="25">
<validator type="regex" message="Please name the feature with only alphanumeric characters.">[a-zA-Z0-9]{0,25}</validator>
</param>
</repeat>
</inputs>
<outputs>
<data format="genetrack" name="genetrack" />
<data format="bed" name="bed_out" />
</outputs>
<help>
This tool takes the input Fit Data and creates a peak and curve plot showing
the reads and fitness on each basepair. Features can be plotted below as tracks.
-----
**Syntax**
- **Track Label** is the name of the generated track.
- **Fit Data** are the datasets to calculate coverage/reads across basepairs and generate a curve.
- **Features** are additional datasets (interval format) to be plotted below as tracks.
</help>
</tool>
+13
View File
@@ -0,0 +1,13 @@
import sets, os
from galaxy import eggs
from galaxy import jobs
from galaxy.tools.parameters import DataToolParameter
def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None):
"""
Copy data_label to genetrack.metadata.label
"""
out_data['genetrack'].metadata.label = param_dict['data_label']
out_data['genetrack'].info = "Use the link below to view the custom track."
out_data['bed_out'].info = ""