Initial GeneTrack commit. Most parts are in, but dependencies will still be a problem.

This commit is contained in:
Ian Schenck
2009-01-19 17:45:28 -05:00
parent 7c57ddc959
commit 6a0aa166aa
13 changed files with 594 additions and 5 deletions
+139
View File
@@ -0,0 +1,139 @@
#!/usr/bin/env python
"""
Run GeneTrack(atlas) with a faked conf file to generate GeneTrack data files.
usage: %prog
-l, --label=N: Data label for fit curve/peak plot
-1, --fits=N/N/N/N/N,...: Data files (interval format) for fit curve/peak plot
-2, --feats=N:M/N/N/N/N/N,...: Data files (interval format) for features.
-d, --data=N: Output path for hdf5 and sqlite databases.
-o, --output=N: Output path for export file.
"""
from galaxy import eggs
import pkg_resources
pkg_resources.require("GeneTrack")
pkg_resources.require("bx-python")
from atlas import commands
from bx.cookbook import doc_optparse
import os
import commands as oscommands
import tempfile
SIGMA = 20
WIDTH = 5 * SIGMA
EXCLUSION_ZONE = 147
def main(label, fit, feats, data_dir, output):
os.mkdir(data_dir)
conf = DummyConf(
__name__=label,
CLOBBER = True,
DATA_SIZE = 3*10**6,
MINIMUM_PEAK_SIZE = 0.1,
LOADER_ENABLED = True,
FITTER_ENABLED = True,
PREDICTOR_ENABLED = True,
EXPORTER_ENABLED = True,
LOADER = loader,
FITTER = fitter,
PREDICTOR = predictor,
EXPORTER = exporter,
HDF_DATABASE = os.path.join( data_dir, "data.hdf" ),
SQL_URI = "sqlite:///%s" % os.path.join( data_dir, "features.sqlite" ),
SIGMA = SIGMA,
WIDTH = WIDTH,
DATA_LABEL = label,
FIT_LABEL = "%s-SIGMA-%d" % ( label,SIGMA ),
PEAK_LABEL = "PRED-%s-SIGMA-%d" % ( label,SIGMA ),
EXCLUSION_ZONE = EXCLUSION_ZONE,
LEFT_SHIFT = EXCLUSION_ZONE / 2,
RIGHT_SHIFT = EXCLUSION_ZONE / 2,
EXPORT_LABELS = [ "PRED-%s-SIGMA-%d" % ( label,SIGMA ) ],
EXPORT_DIR = os.path.join( data_dir ),
DATA_FILE=fit[1],
fit=fit,
feats=feats,
)
commands.execute(conf)
# mod454 seems to be a module without a package. The necessary funcitons are
# stubbed out here until I'm sure of their final home. INS
def loader( conf ):
from atlas import hdf
from mod454.schema import Mod454Schema as Schema
last_chrom = table = None
db = hdf.hdf_open( conf.HDF_DATABASE, mode='a', title='HDF database')
gp = hdf.create_group( db=db, name=conf.DATA_LABEL, desc='data group', clobber=conf.CLOBBER )
fit_meta = conf.fit[2]
# iterate over the file and insert into table
for line in open( conf.fit[1], "r" ):
if line.startswith("chrom"): continue #Skip possible header
if line.startswith("#"): continue
fields = line.rstrip('\r\n').split('\t')
chrom = fields[fit_meta.chromCol]
if chrom != last_chrom:
if table: table.flush()
table = hdf.create_table( db=db, name=chrom, where=gp, schema=Schema, clobber=False )
last_chrom = chrom
try:
position = int(fields[fit_meta.positionCol])
forward = float(fields[fit_meta.forwardCol])
reverse = fit_meta.reverseCol > -1 and float(fields[fit_meta.reverseCol]) or 0.0
row = ( position, forward, reverse, forward+reverse, )
table.append( [ row ] )
except ValueError:
# Ignore bad lines
pass
table.flush()
db.close()
def fitter( conf ):
from mod454.fitter import fitter as mod454_fitter
return mod454_fitter( conf )
def predictor( conf ):
from mod454.predictor import predictor as mod454_predictor
return mod454_predictor( conf )
def exporter( conf ):
return commands.bed_exporter(conf)
class Bunch( object ):
def __init__(self, **kwargs):
for key,value in kwargs.items():
setattr( self, key, value )
class DummyConf( Bunch ):
"""
Fake conf module for genetrack/atlas.
"""
pass
if __name__ == "__main__":
options, args = doc_optparse.parse( __doc__ )
try:
label = options.label
fit_name, fit_meta = options.fits.split(':')[0], [int(x)-1 for x in options.fits.split(':')[1:]]
fit_meta = Bunch(chromCol=fit_meta[0], positionCol=fit_meta[1], forwardCol=fit_meta[2], reverseCol=fit_meta[3])
fit = ( label, fit_name, fit_meta, )
# split apart the string into nested lists, preserves order
if options.feats:
feats = [ (
feat_label,
fname,
Bunch(chromCol=int(chromCol)-1, startCol=int(startCol)-1, endCol=int(endCol)-1,
strandCol=int(strandCol)-1, nameCol=int(nameCol)-1),
)
for feat_label, fname, chromCol, startCol, endCol, strandCol, nameCol
in ( feat.split(':') for feat in options.feats.split(',') )]
else:
feats = []
data_dir = options.data
output = options.output
except:
doc_optparse.exception()
main(label, fit, feats, data_dir, output)
+53
View File
@@ -0,0 +1,53 @@
<tool id="genetrack1" name="GeneTrack">
<description>Track creator/viewer</description>
<code file="genetrack_code.py">
<hook exec_after_process="exec_after_process" />
</code>
<command interpreter="python">
genetrack.py -l $data_label
-1 ${fit_data}:${fit_data.metadata.chromCol}:${fit_data.metadata.positionCol}:${fit_data.metadata.forwardCol}:${fit_data.metadata.reverseCol}
#if $feature_data
-2
#end if
#for $data in $feature_data
${data.name}:${data.input}:${data.input.metadata.chromCol}:${data.input.metadata.startCol}:${data.input.metadata.endCol}:${data.input.metadata.strandCol}:${data.input.metadata.nameCol},
#end for
-d ${genetrack.files_path}
-o ${bed_out}
</command>
<inputs>
<param name="data_label" type="text" label="Track Label" size="50">
<validator type="regex" message="Please name the track with only alphanumeric characters.">[a-zA-Z0-9]{0,25}</validator>
</param>
<param name="fit_data" type="data" format="coverage" label="Coverage Dataset" />
<repeat name="feature_data" title="Features">
<param name="input" type="data" format="interval" label="Dataset" />
<param name="name" type="text" label="Feature Type (mRNA, ESTs, ORFs, etc.)" size="25">
<validator type="regex" message="Please name the feature with only alphanumeric characters.">[a-zA-Z0-9]{0,25}</validator>
</param>
</repeat>
</inputs>
<outputs>
<data format="genetrack" name="genetrack" />
<data format="bed" name="bed_out" />
</outputs>
<help>
This tool takes the input Fit Data and creates a peak and curve plot showing
the reads and fitness on each basepair. Features can be plotted below as tracks.
-----
**Syntax**
- **Track Label** is the name of the generated track.
- **Fit Data** are the datasets to calculate coverage/reads across basepairs and generate a curve.
- **Features** are additional datasets (interval format) to be plotted below as tracks.
</help>
</tool>
+13
View File
@@ -0,0 +1,13 @@
import sets, os
from galaxy import eggs
from galaxy import jobs
from galaxy.tools.parameters import DataToolParameter
def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None):
"""
Copy data_label to genetrack.metadata.label
"""
out_data['genetrack'].metadata.label = param_dict['data_label']
out_data['genetrack'].info = "Use the link below to view the custom track."
out_data['bed_out'].info = ""