Changed GALAXY_DATA_INDEX_DIR from an environment variable to a config entry.

This commit is contained in:
Greg Von Kuster
2008-03-25 14:09:27 +00:00
parent d77480272f
commit 749f218f08
35 changed files with 103 additions and 103 deletions
+2 -2
View File
@@ -32,6 +32,7 @@ class Configuration( object ):
self.file_path = resolve_path( kwargs.get( "file_path", "database/files" ), self.root )
self.new_file_path = resolve_path( kwargs.get( "new_file_path", "database/tmp" ), self.root )
self.tool_path = resolve_path( kwargs.get( "tool_path", "tools" ), self.root )
self.tool_data_path = resolve_path( kwargs.get( "tool_data_path", "tool-data" ), os.getcwd() )
self.test_conf = resolve_path( kwargs.get( "test_conf", "" ), self.root )
self.tool_config = resolve_path( kwargs.get( 'tool_config_file', 'tool_conf.xml' ), self.root )
self.tool_secret = kwargs.get( "tool_secret", "" )
@@ -55,7 +56,6 @@ class Configuration( object ):
self.pbs_stage_path = kwargs.get('pbs_stage_path', "" )
self.use_heartbeat = kwargs.get( 'use_heartbeat', False )
self.ucsc_display_sites = kwargs.get( 'ucsc_display_sites', "main,test,archaea" ).lower().split(",")
self.gbrowse_display_sites = kwargs.get( 'gbrowse_display_sites', "wormbase,flybase" ).lower().split(",")
self.brand = kwargs.get( 'brand', None )
self.wiki_url = kwargs.get( 'wiki_url', None )
self.bugs_email = kwargs.get( 'bugs_email', None )
@@ -87,7 +87,7 @@ class Configuration( object ):
return self.config_dict.get( key, default )
def check( self ):
# Check that required directories exist
for path in self.root, self.file_path, self.tool_path, self.template_path, self.job_working_directory, self.datatype_converters_path:
for path in self.root, self.file_path, self.tool_path, self.tool_data_path, self.template_path, self.job_working_directory, self.datatype_converters_path:
if not os.path.isdir( path ):
raise ConfigurationError("Directory does not exist: %s" % path )
# Check that required files exist
+2 -4
View File
@@ -21,7 +21,6 @@ pbs_template = """#!/bin/sh
export LC_ALL='%s'
export PATH='%s'
export PYTHONPATH='%s'
export GALAXY_DATA_INDEX_DIR='%s'
cd %s
%s
"""
@@ -30,7 +29,6 @@ pbs_symlink_template = """#!/bin/sh
export LC_ALL='%s'
export PATH='%s'
export PYTHONPATH='%s'
export GALAXY_DATA_INDEX_DIR='%s'
for dataset in %s; do
dir=`dirname $dataset`
file=`basename $dataset`
@@ -154,9 +152,9 @@ class PBSJobRunner( object ):
# write the job script
if self.app.config.pbs_stage_path != '':
script = pbs_symlink_template % (os.environ['LC_ALL'], os.environ['NODEPATH'], os.environ['PYTHONPATH'], os.environ['GALAXY_DATA_INDEX_DIR'], " ".join(job_wrapper.get_input_fnames() + job_wrapper.get_output_fnames()), self.app.config.pbs_stage_path, exec_dir, command_line)
script = pbs_symlink_template % (os.environ['LC_ALL'], os.environ['NODEPATH'], os.environ['PYTHONPATH'], " ".join(job_wrapper.get_input_fnames() + job_wrapper.get_output_fnames()), self.app.config.pbs_stage_path, exec_dir, command_line)
else:
script = pbs_template % (os.environ['LC_ALL'], os.environ['NODEPATH'], os.environ['PYTHONPATH'], os.environ['GALAXY_DATA_INDEX_DIR'], exec_dir, command_line)
script = pbs_template % (os.environ['LC_ALL'], os.environ['NODEPATH'], os.environ['PYTHONPATH'], exec_dir, command_line)
job_file = "%s/database/pbs/%s.sh" % (os.getcwd(), job_wrapper.job_id)
fh = file(job_file, "w")
fh.write(script)
+1 -1
View File
@@ -991,7 +991,7 @@ class Tool:
# TODO: path munging for cluster/dataset server relocatability
param_dict['__new_file_path__'] = os.path.abspath(self.app.config.new_file_path)
# The following points to location (xxx.loc) files which are pointers to locally cached data
param_dict['GALAXY_DATA_INDEX_DIR'] = os.environ.get( 'GALAXY_DATA_INDEX_DIR' )
param_dict['GALAXY_DATA_INDEX_DIR'] = self.app.config.tool_data_path
# Return the dictionary of parameters
return param_dict
+25 -27
View File
@@ -14,20 +14,14 @@ class DynamicOptions( object ):
self.parameter_type = parameter_type
self.data_ref = None
self.param_ref = None
self.from_file_data = None
self.data_file_path = None
self.data_file_data = None
self.validators = []
# Parse the options tag
self.data_file = elem.get( 'from_file', None )
if self.data_file is not None:
self.data_file = self.data_file.strip()
if self.data_file.startswith( 'static' ):
# static files ( ucsc/builds.txt, etc ) have relative paths in the tool config
self.from_file = self.data_file
else:
self.from_file = "%s/%s" % ( os.environ.get( 'GALAXY_DATA_INDEX_DIR' ), self.data_file )
else:
self.from_file = None
self.name_col = elem.get( 'name_col', None )
if self.name_col is not None:
self.name_col = int( self.name_col.strip() )
@@ -119,8 +113,8 @@ class DynamicOptions( object ):
elif self.data_file == 'data_ref':
# We'll be reading data directly from the input dataset
filters[ 'data_meta' ][ 'meta_key' ] = 'data_ref'
self.from_file = dataset.get_file_name()
filters[ 'data_meta' ][ 'meta_value' ] = self.from_file
self.data_file_path = dataset.get_file_name()
filters[ 'data_meta' ][ 'meta_value' ] = self.data_file_path
# meta_key_col is optional
meta_key_col = filter.get( 'meta_key_col', None )
if meta_key_col is not None:
@@ -156,6 +150,10 @@ class DynamicOptions( object ):
except:
filters[ 'params' ] = {}
filters[ 'params' ][ n ] = v
if self.data_file and self.data_file.startswith( 'static' ):
self.data_file_path = self.data_file
elif self.data_file not in [ None, 'data_ref' ]:
self.data_file_path = "%s/%s" % ( trans.app.config.tool_data_path, self.data_file )
# Now that we've parsed our filters, we need to see if the tool is a maf tool
# which requires special handling
# TODO: remove or rework this if possible
@@ -200,7 +198,7 @@ class DynamicOptions( object ):
dbkey = filters[ 'params' ][ 'dbkey' ]
return self.generate_for_encode( encode_group, dbkey, sep )
elif self.data_file == 'microbial_data.loc':
if self.from_file_data is None:
if self.data_file_data is None:
self.load_microbial_data()
try:
kingdom = filters[ 'param_values' ][ 'kingdom' ]
@@ -232,7 +230,7 @@ class DynamicOptions( object ):
options = []
def generate():
encode_sets = {}
for line in open( self.from_file ):
for line in open( self.data_file_path ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ):
try:
@@ -313,42 +311,42 @@ class DynamicOptions( object ):
def generate_for_microbial( self, kingdom=None, org=None, feature=None ):
options = []
if not kingdom and not org and not feature:
kingdoms = self.from_file_data.keys()
kingdoms = self.data_file_data.keys()
kingdoms.sort()
for kingdom in kingdoms:
options.append( ( kingdom, kingdom, False ) )
if options:
options[0] = ( options[0][0], options[0][1], True )
elif kingdom and not org and not feature:
orgs = self.from_file_data[ kingdom ].keys()
orgs = self.data_file_data[ kingdom ].keys()
#need to sort by name
swap_test = False
for i in range( 0, len( orgs ) - 1 ):
for j in range( 0, len( orgs ) - i - 1 ):
if self.from_file_data[ kingdom ][ orgs[ j ] ][ 'name' ] > self.from_file_data[ kingdom ][ orgs[ j + 1 ] ][ 'name' ]:
if self.data_file_data[ kingdom ][ orgs[ j ] ][ 'name' ] > self.data_file_data[ kingdom ][ orgs[ j + 1 ] ][ 'name' ]:
orgs[ j ], orgs[ j + 1 ] = orgs[ j + 1 ], orgs[ j ]
swap_test = True
if swap_test == False:
break
for org in orgs:
if self.from_file_data[ kingdom ][ org ][ 'link_site' ] == "UCSC":
options.append( ( "<b>" + self.from_file_data[ kingdom ][ org ][ 'name' ] + "</b> <a href=\"" + self.from_file_data[ kingdom ][ org ][ 'info_url' ] + "\" target=\"_blank\">(about)</a>", org, False ) )
if self.data_file_data[ kingdom ][ org ][ 'link_site' ] == "UCSC":
options.append( ( "<b>" + self.data_file_data[ kingdom ][ org ][ 'name' ] + "</b> <a href=\"" + self.data_file_data[ kingdom ][ org ][ 'info_url' ] + "\" target=\"_blank\">(about)</a>", org, False ) )
else:
options.append( ( self.from_file_data[ kingdom ][ org ][ 'name' ] + " <a href=\"" + self.from_file_data[ kingdom ][ org ][ 'info_url' ] + "\" target=\"_blank\">(about)</a>", org, False ) )
options.append( ( self.data_file_data[ kingdom ][ org ][ 'name' ] + " <a href=\"" + self.data_file_data[ kingdom ][ org ][ 'info_url' ] + "\" target=\"_blank\">(about)</a>", org, False ) )
if options:
options[0] = ( options[0][0], options[0][1], True )
else:
chroms = self.from_file_data[ kingdom ][ org ][ 'chrs' ].keys()
chroms = self.data_file_data[ kingdom ][ org ][ 'chrs' ].keys()
chroms.sort()
for chr in chroms:
for data in self.from_file_data[ kingdom ][ org ][ 'chrs' ][ chr ][ 'data' ]:
if self.from_file_data[ kingdom ][ org ][ 'chrs' ][ chr ][ 'data' ][ data ][ 'feature' ] == feature:
options.append( ( self.from_file_data[ kingdom ][ org ][ 'chrs' ][ chr ][ 'name' ] + " <a href=\"" + self.from_file_data[ kingdom ][ org ][ 'chrs' ][ chr ][ 'info_url' ] + "\" target=\"_blank\">(about)</a>", data, False ) )
for data in self.data_file_data[ kingdom ][ org ][ 'chrs' ][ chr ][ 'data' ]:
if self.data_file_data[ kingdom ][ org ][ 'chrs' ][ chr ][ 'data' ][ data ][ 'feature' ] == feature:
options.append( ( self.data_file_data[ kingdom ][ org ][ 'chrs' ][ chr ][ 'name' ] + " <a href=\"" + self.data_file_data[ kingdom ][ org ][ 'chrs' ][ chr ][ 'info_url' ] + "\" target=\"_blank\">(about)</a>", data, False ) )
return options
def load_microbial_data( self, sep='\t' ):
microbe_info= {}
orgs = {}
for line in open( self.from_file ):
for line in open( self.data_file_path ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ):
fields = line.split( sep )
@@ -421,7 +419,7 @@ class DynamicOptions( object ):
microbe_info[ org[ 'kingdom' ] ] = {}
if org_num not in microbe_info[ org[ 'kingdom' ] ]:
microbe_info[ org[ 'kingdom' ] ][org_num] = org
self.from_file_data = microbe_info
self.data_file_data = microbe_info
def generate_for_species( self, species ):
options = []
for s in species:
@@ -430,7 +428,7 @@ class DynamicOptions( object ):
def generate_for_maf( self, maf_uid, sep ):
options = []
d = {}
for line in open( self.from_file ):
for line in open( self.data_file_path ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ):
fields = line.split( sep )
@@ -476,7 +474,7 @@ class DynamicOptions( object ):
def generate_for_build( self, build, build_col, name_col, value_col, sep ):
options = []
d = {}
for line in open( self.from_file ):
for line in open( self.data_file_path ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ):
fields = line.split( sep )
@@ -528,7 +526,7 @@ class DynamicOptions( object ):
return options
def generate( self, name_col, value_col, sep ):
options = []
for line in open( self.from_file ):
for line in open( self.data_file_path ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ):
fields = line.split( sep )
+8 -2
View File
@@ -212,17 +212,23 @@ class MetadataInFileColumnValidator( Validator ):
@classmethod
def from_element( cls, elem ):
filename = elem.get( "filename", None )
if filename:
filename = filename.strip()
metadata_name = elem.get( "metadata_name", None )
if metadata_name:
metadata_name = metadata_name.strip()
metadata_column = int( elem.get( "metadata_column", 0 ) )
message = elem.get( "message", "Value for metadata %s was not found in %s." % ( metadata_name, filename ) )
split = elem.get( "split", None )
line_startswith = elem.get( "line_startswith", None )
if line_startswith:
line_startswith = line_startswith.strip()
return cls( filename, metadata_name, metadata_column, message, split, line_startswith )
def __init__( self, filename, metadata_name, metadata_column, message = "Value for metadata not found." , split = None, line_startswith = None ):
def __init__( self, filename, metadata_name, metadata_column, message="Value for metadata not found.", split=None, line_startswith=None ):
self.metadata_name = metadata_name
self.message = message
self.valid_values = []
filename = "%s/%s" % ( os.environ.get( 'GALAXY_DATA_INDEX_DIR' ), filename )
#filename = "%s/%s" % ( GALAXY_DATA_INDEX_DIR, filename )
for line in open( filename ):
if line_startswith is None or line.startswith( line_startswith ):
fields = line.split( split )
+2 -4
View File
@@ -9,8 +9,6 @@ import bx.intervals
import bx.interval_index_file
import sys, os, string, tempfile
MAF_LOCATION_FILE = "%s/maf_index.loc" % os.environ.get( 'GALAXY_DATA_INDEX_DIR' )
#an object corresponding to a reference layered alignment
class RegionAlignment( object ):
@@ -126,8 +124,8 @@ class SplicedAlignment( object ):
return self.exons[-1].end
#Open a MAF index using a UID
def maf_index_by_uid( maf_uid, index_location_file = None ):
for line in open( index_location_file or MAF_LOCATION_FILE ):
def maf_index_by_uid( maf_uid, index_location_file ):
for line in open( index_location_file ):
try:
#read each line, if not enough fields, go to next line
if line[0:1] == "#" : continue
+1 -6
View File
@@ -14,12 +14,7 @@ echo "Architecture appears to be $ARCH $PYTHON_UCS"
UNIVERSE_HOME=`pwd`
PYTHONPATH=$UNIVERSE_HOME/lib:$UNIVERSE_HOME/eggs/$PYTHON_UCS:$UNIVERSE_HOME/eggs
# The following directory must include location ( xxx.loc ) files that point to the
# location where the locally cached data is stored. These location files are used
# by various tools.
GALAXY_DATA_INDEX_DIR=$UNIVERSE_HOME/tool-data
export UNIVERSE_HOME PYTHONPATH GALAXY_DATA_INDEX_DIR
export UNIVERSE_HOME PYTHONPATH
## For PBS - if you need to force node paths (i.e. the arch the frontend
## runs on is not the same arch as the compute nodes or Galaxy is
+1 -1
View File
@@ -16,7 +16,7 @@ def stop_err( msg ):
def main():
uids = sys.argv[1].split(",")
out_file1 = sys.argv[2]
file_name = "%s/encode_datasets.loc" % os.environ.get( 'GALAXY_DATA_INDEX_DIR' )
file_name = "%s/encode_datasets.loc" % sys.argv[3]
#remove NONE from uids
have_none = True
@@ -1,5 +1,5 @@
<tool id="encode_import_all_latest_datasets1" name="Combined Datasets">
<command interpreter="python">encode_import.py $hg17,$hg16 $output</command>
<command interpreter="python">encode_import.py $hg17,$hg16 $output ${GALAXY_DATA_INDEX_DIR}</command>
<inputs>
<display>
<p><div class="toolFormTitle">hg1 (most recent datasets in bold)</div>$hg17</p>
@@ -1,5 +1,5 @@
<tool id="encode_import_chromatin_and_chromosomes1" name="Chromatin and Chromosomes">
<command interpreter="python">encode_import.py $hg17,$hg16 $output</command>
<command interpreter="python">encode_import.py $hg17,$hg16 $output ${GALAXY_DATA_INDEX_DIR}</command>
<inputs>
<display>
<p><div class="toolFormTitle">hg17 (most recent datasets in bold)</div>$hg17</p>
+1 -1
View File
@@ -1,5 +1,5 @@
<tool id="encode_import_gencode1" name="Gencode Datasets">
<command interpreter="python">encode_import.py $hg17,$hg16 $output</command>
<command interpreter="python">encode_import.py $hg17,$hg16 $output ${GALAXY_DATA_INDEX_DIR}</command>
<inputs>
<display>
<p><div class="toolFormTitle">hg17 (most recent datasets in bold)</div>$hg17</p>
@@ -1,5 +1,5 @@
<tool id="encode_import_genes_and_transcripts1" name="Genes and Transcripts">
<command interpreter="python">encode_import.py $hg17,$hg16 $output</command>
<command interpreter="python">encode_import.py $hg17,$hg16 $output ${GALAXY_DATA_INDEX_DIR}</command>
<inputs>
<display>
<p><div class="toolFormTitle">hg17 (most recent datasets in bold)</div>$hg17</p>
@@ -1,5 +1,5 @@
<tool id="encode_import_multi-species_sequence_analysis1" name="Multi-species Sequence Analysis">
<command interpreter="python">encode_import.py $hg17,$hg16 $output</command>
<command interpreter="python">encode_import.py $hg17,$hg16 $output ${GALAXY_DATA_INDEX_DIR}</command>
<inputs>
<display>
<p><div class="toolFormTitle">hg17 (most recent datasets in bold)</div>$hg17</p>
@@ -1,5 +1,5 @@
<tool id="encode_import_transcription_regulation1" name="Transcription Regulation">
<command interpreter="python">encode_import.py $hg17,$hg16 $output</command>
<command interpreter="python">encode_import.py $hg17,$hg16 $output ${GALAXY_DATA_INDEX_DIR}</command>
<inputs>
<display>
<p><div class="toolFormTitle">hg17 (most recent datasets in bold)</div>$hg17</p>
+1 -1
View File
@@ -24,7 +24,7 @@ while have_none:
#create dictionary keyed by uid of tuples of (displayName,filePath,build) for all files
available_files = {}
try:
filename = "%s/microbial_data.loc" % os.environ.get( 'GALAXY_DATA_INDEX_DIR' )
filename = sys.argv[9]
for i, line in enumerate( file( filename ) ):
if not line or line[0:1] == "#" : continue
fields = line.split('\t')
+1 -1
View File
@@ -1,5 +1,5 @@
<tool id="microbial_import1" name="Get Microbial Data">
<command interpreter="python">microbial_import.py $CDS,$tRNA,$rRNA,$sequence,$GeneMark,$GeneMarkHMM,$Glimmer3 $output</command>
<command interpreter="python">microbial_import.py $CDS,$tRNA,$rRNA,$sequence,$GeneMark,$GeneMarkHMM,$Glimmer3 $output ${GALAXY_DATA_INDEX_DIR}/microbial_data.loc</command>
<inputs>
<page>
<display>
+3 -3
View File
@@ -1,11 +1,11 @@
def load_microbial_data( sep='\t' ):
def load_microbial_data( GALAXY_INDEX_DATA_DIR, sep='\t' ):
# FIXME: this function is duplicated in the DynamicOptions class. It is used here only to
# set data.name in exec_after_process().
microbe_info= {}
orgs = {}
filename = "%s/microbial_data.loc" % os.environ.get( 'GALAXY_DATA_INDEX_DIR' )
filename = "%s/microbial_data.loc" % GALAXY_DATA_INDEX_DIR
for i, line in enumerate( file( filename ) ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ):
@@ -96,7 +96,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr
if not (kingdom or org):
print "Parameters are not available."
microbe_info = load_microbial_data()
microbe_info = load_microbial_data( param_dict.get( 'GALAXY_INDEX_DATA_DIR' ), sep='\t' )
new_stdout = ""
split_stdout = stdout.split("\n")
basic_name = ""
+1 -1
View File
@@ -1,6 +1,6 @@
<tool id="random_intervals1" name="Random Intervals">
<description>create a random set of intervals</description>
<command interpreter="python2.4">random_intervals_no_bits.py $regions $input2 $input1 $out_file1 $input2_chromCol $input2_startCol $input2_endCol $input1_chromCol $input1_startCol $input1_endCol $input1_strandCol $use_mask $strand_overlaps</command>
<command interpreter="python2.4">random_intervals_no_bits.py $regions $input2 $input1 $out_file1 $input2_chromCol $input2_startCol $input2_endCol $input1_chromCol $input1_startCol $input1_endCol $input1_strandCol $use_mask $strand_overlaps ${GALAXY_DATA_INDEX_DIR}</command>
<inputs>
<param name="input1" type="data" format="interval" label="File to Mimick">
<validator type="unspecified_build" message="Unspecified build, this tool works with data from genome builds hg16 or hg17. Click the pencil icon in your history item to set the genome build."/>
+1 -1
View File
@@ -106,7 +106,7 @@ def main():
use_mask = sys.argv[11]
overlaps = sys.argv[12]
available_regions = {}
loc_file = "%s/regions.loc" % os.environ.get( 'GALAXY_DATA_INDEX_DIR' )
loc_file = "%s/regions.loc" % sys.argv[-1]
for i, line in enumerate( file( loc_file ) ):
line = line.rstrip( '\r\n' )
+8 -8
View File
@@ -1,6 +1,6 @@
#!/usr/bin/env python2.4
"""
usage: extract_genomic_dna.py $input $out_file1 $input_chromCol $input_startCol $input_endCol $input_strandCol $dbkey $out_format
usage: extract_genomic_dna.py $input $out_file1 $input_chromCol $input_startCol $input_endCol $input_strandCol $dbkey $out_format GALAXY_DATA_INDEX_DIR
by Wen-Yu Chung
"""
import pkg_resources
@@ -10,9 +10,6 @@ from bx.cookbook import doc_optparse
import bx.seq.nib
import bx.seq.twobit
nib_file = "%s/alignseq.loc" % os.environ.get( 'GALAXY_DATA_INDEX_DIR' )
twobit_file = "%s/twobit.loc" % os.environ.get( 'GALAXY_DATA_INDEX_DIR' )
def stop_err( msg ):
sys.stderr.write( msg )
sys.exit()
@@ -25,7 +22,8 @@ def reverse_complement( s ):
reversed_s.reverse()
return "".join( reversed_s )
def check_nib_file( dbkey ):
def check_nib_file( dbkey, GALAXY_DATA_INDEX_DIR ):
nib_file = "%s/alignseq.loc" % GALAXY_DATA_INDEX_DIR
nib_path = ''
nibs = {}
for line in open( nib_file ):
@@ -40,7 +38,8 @@ def check_nib_file( dbkey ):
nib_path = nibs[( dbkey )]
return nib_path
def check_twobit_file( dbkey ):
def check_twobit_file( dbkey, GALAXY_DATA_INDEX_DIR ):
twobit_file = "%s/twobit.loc" % GALAXY_DATA_INDEX_DIR
twobit_path = ''
twobits = {}
for line in open( twobit_file ):
@@ -80,10 +79,11 @@ def __main__():
pass
dbkey = sys.argv[7]
output_format = sys.argv[8]
GALAXY_DATA_INDEX_DIR = sys.argv[9]
nibs = {}
twobits = {}
nib_path = check_nib_file( dbkey )
twobit_path = check_twobit_file( dbkey )
nib_path = check_nib_file( dbkey, GALAXY_DATA_INDEX_DIR )
twobit_path = check_twobit_file( dbkey, GALAXY_DATA_INDEX_DIR )
if not os.path.exists( nib_path ) and not os.path.exists( twobit_path ):
# If this occurs, we need to fix the metadata validator.
stop_err( "No sequences are available for '%s', request them by reporting this error." % dbkey )
+2 -2
View File
@@ -1,10 +1,10 @@
<tool id="Extract genomic DNA 1" name="Extract Genomic DNA" version="2.0.0">
<description>using coordinates from assembled/unassebmled genomes</description>
<command interpreter="python">extract_genomic_dna.py $input $out_file1 $input_chromCol $input_startCol $input_endCol $input_strandCol $dbkey $out_format</command>
<command interpreter="python">extract_genomic_dna.py $input $out_file1 $input_chromCol $input_startCol $input_endCol $input_strandCol $dbkey $out_format ${GALAXY_DATA_INDEX_DIR}</command>
<inputs>
<param format="interval" name="input" type="data" label="Fetch sequences corresponding to Query">
<validator type="unspecified_build" />
<validator type="dataset_metadata_in_file" filename="alignseq.loc" metadata_name="dbkey" metadata_column="1" message="Sequences are not currently available for the specified build." split=" " line_startswith="seq" />
<validator type="dataset_metadata_in_file" filename="/depot/data2/galaxy/alignseq.loc" metadata_name="dbkey" metadata_column="1" message="Sequences are not currently available for the specified build." split=" " line_startswith="seq" />
</param>
<param name="out_format" type="select" label="Output data type">
<option value="fasta">FASTA</option>
-1
View File
@@ -1,6 +1,5 @@
<tool id="axt_to_lav_1" name="AXT to LAV">
<description>Converts an AXT formated file to LAV format</description>
<!-- <command interpreter="python2.4">axt_to_lav.py $align_input $dbkey_1 $dbkey_2 $lav_file $seq_file1 $seq_file2</command> -->
<command interpreter="python2.4">axt_to_lav.py ${GALAXY_DATA_INDEX_DIR}/$dbkey_1/seq/%s.nib:$dbkey_1:./static/ucsc/chrom/${dbkey_1}.len ${GALAXY_DATA_INDEX_DIR}/$dbkey_2/seq/%s.nib:$dbkey_2:./static/ucsc/chrom/${dbkey_2}.len $align_input $lav_file $seq_file1 $seq_file2</command>
<inputs>
<param name="align_input" type="data" format="axt" label="Alignment File" optional="False"/>
+2 -2
View File
@@ -1,7 +1,7 @@
<tool id="GeneBed_Maf_Fasta2" name="Stitch Gene blocks">
<description>given a set of coding exon intervals</description>
<command interpreter="python2.4">#if $maf_source_type.maf_source == "user":#interval_maf_to_merged_fasta.py --dbkey=$dbkey --species=$maf_source_type.species --mafSource=$maf_source_type.maf_file --interval_file=$input1 --output_file=$out_file1 --mafSourceType=$maf_source_type.maf_source --geneBED
#else:#interval_maf_to_merged_fasta.py --dbkey=$dbkey --species=$maf_source_type.species --mafSource=$maf_source_type.maf_identifier --interval_file=$input1 --output_file=$out_file1 --mafSourceType=$maf_source_type.maf_source --geneBED
<command interpreter="python2.4">#if $maf_source_type.maf_source == "user":#interval_maf_to_merged_fasta.py --dbkey=$dbkey --species=$maf_source_type.species --mafSource=$maf_source_type.maf_file --interval_file=$input1 --output_file=$out_file1 --mafSourceType=$maf_source_type.maf_source --geneBED --mafIndexFileDir=${GALAXY_DATA_INDEX_DIR}
#else:#interval_maf_to_merged_fasta.py --dbkey=$dbkey --species=$maf_source_type.species --mafSource=$maf_source_type.maf_identifier --interval_file=$input1 --output_file=$out_file1 --mafSourceType=$maf_source_type.maf_source --geneBED --mafIndexFileDir=${GALAXY_DATA_INDEX_DIR}
#end if
</command>
<inputs>
+2 -1
View File
@@ -20,6 +20,7 @@ usage: %prog maf_file [options]
-o, --output_file=o: Output MAF file
-p, --species=p: Species to include in output
-l, --indexLocation=l: Override default maf_index.loc file
-z, --mafIndexFile=z: Directory of local maf index file ( maf_index.loc or maf_pairwise.loc )
"""
#Dan Blankenberg
@@ -84,7 +85,7 @@ def __main__():
if options.indexLocation:
index = maf_utilities.maf_index_by_uid( options.mafType, options.indexLocation )
else:
index = maf_utilities.maf_index_by_uid( options.mafType )
index = maf_utilities.maf_index_by_uid( options.mafType, options.mafIndexFile )
if index is None:
print >> sys.stderr, "The MAF source specified (%s) appears to be invalid." % ( options.mafType )
sys.exit()
+2 -2
View File
@@ -1,8 +1,8 @@
<tool id="Interval2Maf1" name="Extract MAF blocks">
<description>given a set of genomic intervals</description>
<command interpreter="python2.4">
#if $maf_source_type.maf_source == "user":#interval2maf.py --dbkey=$input1_dbkey --chromCol=$input1_chromCol --startCol=$input1_startCol --endCol=$input1_endCol --strandCol=$input1_strandCol --mafFile=$maf_source_type.mafFile --interval_file=$input1 --output_file=$out_file1
#else:#interval2maf.py --dbkey=$input1_dbkey --chromCol=$input1_chromCol --startCol=$input1_startCol --endCol=$input1_endCol --strandCol=$input1_strandCol --mafType=$maf_source_type.mafType --interval_file=$input1 --output_file=$out_file1
#if $maf_source_type.maf_source == "user":#interval2maf.py --dbkey=$input1_dbkey --chromCol=$input1_chromCol --startCol=$input1_startCol --endCol=$input1_endCol --strandCol=$input1_strandCol --mafFile=$maf_source_type.mafFile --interval_file=$input1 --output_file=$out_file1 --mafIndexFile=${GALAXY_DATA_INDEX_DIR}/maf_index.loc
#else:#interval2maf.py --dbkey=$input1_dbkey --chromCol=$input1_chromCol --startCol=$input1_startCol --endCol=$input1_endCol --strandCol=$input1_strandCol --mafType=$maf_source_type.mafType --interval_file=$input1 --output_file=$out_file1 --mafIndexFile=${GALAXY_DATA_INDEX_DIR}/maf_index.loc
#end if
</command>
<inputs>
+4 -2
View File
@@ -18,8 +18,9 @@ usage: %prog maf_file [options]
-i, --interval_file=i: Input interval file
-o, --output_file=o: Output MAF file
-p, --species=p: Species to include in output
-z, --mafIndexFileDir=z: Directory of local maf_index.loc file
usage: %prog dbkey_of_BED comma_separated_list_of_additional_dbkeys_to_extract comma_separated_list_of_indexed_maf_files input_gene_bed_file output_fasta_file cached|user
usage: %prog dbkey_of_BED comma_separated_list_of_additional_dbkeys_to_extract comma_separated_list_of_indexed_maf_files input_gene_bed_file output_fasta_file cached|user GALAXY_DATA_INDEX_DIR
"""
#Dan Blankenberg
@@ -82,13 +83,14 @@ def __main__():
else:
print >>sys.stderr, "Strand column has not been specified."
sys.exit()
mafIndexFile = "%s/maf_index.loc" % options.mafIndexFileDir
#Finish parsing command line
#get index for mafs based on type
index = index_filename = None
#using specified uid for locally cached
if options.mafSourceType.lower() in ["cached"]:
index = maf_utilities.maf_index_by_uid( options.mafSource )
index = maf_utilities.maf_index_by_uid( options.mafSource, mafIndexFile )
if index is None:
print >> sys.stderr, "The MAF source specified (%s) appears to be invalid." % ( options.mafSource )
sys.exit()
+2 -2
View File
@@ -1,7 +1,7 @@
<tool id="Interval_Maf_Merged_Fasta2" name="Stitch MAF blocks">
<description>given a set of genomic intervals</description>
<command interpreter="python2.4">#if $maf_source_type.maf_source == "user":#interval_maf_to_merged_fasta.py --dbkey=$dbkey --species=$maf_source_type.species --mafSource=$maf_source_type.maf_file --interval_file=$input1 --output_file=$out_file1 --chromCol=$input1_chromCol --startCol=$input1_startCol --endCol=$input1_endCol --strandCol=$input1_strandCol --mafSourceType=$maf_source_type.maf_source
#else:#interval_maf_to_merged_fasta.py --dbkey=$dbkey --species=$maf_source_type.species --mafSource=$maf_source_type.maf_identifier --interval_file=$input1 --output_file=$out_file1 --chromCol=$input1_chromCol --startCol=$input1_startCol --endCol=$input1_endCol --strandCol=$input1_strandCol --mafSourceType=$maf_source_type.maf_source
<command interpreter="python2.4">#if $maf_source_type.maf_source == "user":#interval_maf_to_merged_fasta.py --dbkey=$dbkey --species=$maf_source_type.species --mafSource=$maf_source_type.maf_file --interval_file=$input1 --output_file=$out_file1 --chromCol=$input1_chromCol --startCol=$input1_startCol --endCol=$input1_endCol --strandCol=$input1_strandCol --mafSourceType=$maf_source_type.maf_source --mafIndexFileDir=${GALAXY_DATA_INDEX_DIR}
#else:#interval_maf_to_merged_fasta.py --dbkey=$dbkey --species=$maf_source_type.species --mafSource=$maf_source_type.maf_identifier --interval_file=$input1 --output_file=$out_file1 --chromCol=$input1_chromCol --startCol=$input1_startCol --endCol=$input1_endCol --strandCol=$input1_strandCol --mafSourceType=$maf_source_type.maf_source --mafIndexFileDir=${GALAXY_DATA_INDEX_DIR}
#end if
</command>
<inputs>
+3 -2
View File
@@ -26,7 +26,8 @@ def __main__():
summary = sys.argv[8].strip()
if summary.lower() == "true": summary = True
else: summary = False
mafIndexFile = "%s/maf_index.loc" % sys.argv[9]
index = index_filename = None
if maf_source_type == "user":
#index maf for use here
@@ -36,7 +37,7 @@ def __main__():
sys.exit()
elif maf_source_type == "cached":
#access existing indexes
index = maf_utilities.maf_index_by_uid( input_maf_filename )
index = maf_utilities.maf_index_by_uid( input_maf_filename, mafIndexFile )
if index is None:
print >> sys.stderr, "The MAF source specified (%s) appears to be invalid." % ( input_maf_filename )
sys.exit()
+1
View File
@@ -7,6 +7,7 @@
#else:
$maf_source_type.maf_source $maf_source_type.mafType $input1 $out_file1 $dbkey $input1_chromCol $input1_startCol $input1_endCol $summary
#end if
${GALAXY_DATA_INDEX_DIR}
</command>
<inputs>
<param format="interval" name="input1" label="Interval File" type="data">
+3 -3
View File
@@ -1,10 +1,10 @@
import os
def load_maf_data( sep='\t' ):
def load_maf_data( GALAXY_DATA_INDEX_DIR, sep='\t' ):
# FIXME: this function is duplicated in the DynamicOptions class. It is used here only to
# set data.name in exec_before_job().
maf_sets = {}
filename = "%s/maf_index.loc" % os.environ.get( 'GALAXY_DATA_INDEX_DIR' )
filename = "%s/maf_index.loc" % GALAXY_DATA_INDEX_DIR
for i, line in enumerate( file( filename ) ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ):
@@ -27,7 +27,7 @@ def load_maf_data( sep='\t' ):
continue
return maf_sets
def exec_before_job(app, inp_data, out_data, param_dict, tool):
maf_sets = load_maf_data()
maf_sets = load_maf_data( app.config.tool_data_path, sep='\t' )
if param_dict[ 'maf_source_type' ][ 'maf_source' ] == "cached":
for name, data in out_data.items():
try:
+8 -8
View File
@@ -2,10 +2,8 @@
import os, sys, tempfile
nib_file = "%s/alignseq.loc" % os.environ.get( 'GALAXY_DATA_INDEX_DIR' )
twobit_file = "/%s/twobit.loc" % os.environ.get( 'GALAXY_DATA_INDEX_DIR' )
def check_nib_file( dbkey ):
def check_nib_file( dbkey, GALAXY_DATA_INDEX_DIR ):
nib_file = "%s/alignseq.loc" % GALAXY_DATA_INDEX_DIR
nib_path = ''
nibs = {}
for i, line in enumerate( file( nib_file ) ):
@@ -20,7 +18,8 @@ def check_nib_file( dbkey ):
nib_path = nibs[( dbkey )]
return nib_path
def check_twobit_file( dbkey ):
def check_twobit_file( dbkey, GALAXY_DATA_INDEX_DIR ):
twobit_file = "%s/twobit.loc" % GALAXY_DATA_INDEX_DIR
twobit_path = ''
twobits = {}
for i, line in enumerate( file( twobit_file ) ):
@@ -43,7 +42,8 @@ def __main__():
min_iden = sys.argv[5]
tile_size = sys.argv[6]
one_off = sys.argv[7]
GALAXY_DATA_INDEX_DIR = sys.argv[8]
all_files = []
if (source_format == '0'):
# check target genome
@@ -51,8 +51,8 @@ def __main__():
if dbkey == '?':
print >> sys.stdout, "No genome build specified. please check your dataset."
sys.exit()
nib_path = check_nib_file( dbkey )
twobit_path = check_twobit_file( dbkey )
nib_path = check_nib_file( dbkey, GALAXY_DATA_INDEX_DIR )
twobit_path = check_twobit_file( dbkey, GALAXY_DATA_INDEX_DIR )
if not os.path.exists( nib_path ) and not os.path.exists( twobit_path ):
print >> sys.stdout, "No sequences are available for %s, request them by reporting this error." % dbkey
sys.exit()
+7 -6
View File
@@ -1,10 +1,11 @@
<tool id="blat_wrapper" name="Run Sequence Alignment" version="1.0.0">
<description>on short reads against genome builds</description>
<command interpreter="python">
#if $source.source_select=="database":#blat_wrapper.py 0 $source.dbkey $input_query $output1 $iden $tile_size $one_off
#else:#blat_wrapper.py 1 $source.input_target $input_query $output1 $iden $tile_size $one_off
#end if
</command>
<description>on short reads against genome builds</description>
<command interpreter="python">
#if $source.source_select=="database":#blat_wrapper.py 0 $source.dbkey $input_query $output1 $iden $tile_size $one_off
#else:#blat_wrapper.py 1 $source.input_target $input_query $output1 $iden $tile_size $one_off
#end if
${GALAXY_DATA_INDEX_DIR}
</command>
<inputs>
<conditional name="source">
<param name="source_select" type="select" label="Target Source">
+2 -3
View File
@@ -6,8 +6,6 @@ run megablast for metagenomics data
import sys, os, tempfile, subprocess
#from megablast_xml_parser import *
DB_LOC = "%s/blastdb.loc" % os.environ.get( 'GALAXY_DATA_INDEX_DIR' )
def __main__():
# file I/O
db_build = sys.argv[1]
@@ -20,7 +18,8 @@ def __main__():
mega_disc_word = sys.argv[6] # -t
mega_disc_type = sys.argv[7] # -N
mega_filter = sys.argv[8] # -F
GALAXY_DATA_INDEX_DIR = sys.argv[9]
DB_LOC = "%s/blastdb.loc" % GALAXY_DATA_INDEX_DIR
output_file = open(output_filename, 'w')
# prepare the database
+1 -1
View File
@@ -1,6 +1,6 @@
<tool id="megablast_wrapper" name="Megablast" version="2.0.0">
<description>for Metagenomics Projects</description>
<command interpreter="python">megablast_wrapper.py $source_select $input_query $output1 $word_size $iden_cutoff $disc_word $disc_type $filter_query</command>
<command interpreter="python">megablast_wrapper.py $source_select $input_query $output1 $word_size $iden_cutoff $disc_word $disc_type $filter_query ${GALAXY_DATA_INDEX_DIR}</command>
<inputs>
<param name="input_query" type="data" format="fasta" label="Query Sequence"/>
<param name="source_select" type="select" display="radio" label="Target database">
+1
View File
@@ -48,6 +48,7 @@ new_file_path = database/tmp
# Tools
tool_config_file = tool_conf.xml
tool_path = tools
tool_data_path = tool-data
# Datatype converters
datatype_converters_config_file = datatype_converters_conf.xml