Move ucsc and gbrowse data from static to tool-data/shared.

Cron scripts on test and main will need to be updated when the server is.

GOPS complement tool and annotation profiler will now correctly use chromosome lengths files if available.
This commit is contained in:
Daniel Blankenberg
2008-06-03 18:53:26 +00:00
parent f28fe5c4f0
commit 86f0862aa4
12 changed files with 59 additions and 37 deletions
+13 -13
View File
@@ -1,6 +1,6 @@
#!/bin/sh
#
# Script to update UCSC static tables. The idea is to update, but if
# Script to update UCSC shared data tables. The idea is to update, but if
# the update fails, not replace current data/tables with error
# messages.
@@ -9,49 +9,49 @@ GALAXY=/galaxy/path
export PYTHONPATH=${GALAXY}/modules:${GALAXY}/eggs:${GALAXY}/lib
# setup directories
mkdir ${GALAXY}/static/ucsc/new
mkdir ${GALAXY}/static/ucsc/chrom/new
mkdir ${GALAXY}/tool-data/shared/ucsc/new
mkdir ${GALAXY}/tool-data/shared/ucsc/chrom/new
date
echo "Updating UCSC static tables."
echo "Updating UCSC shared data tables."
# Try to build "builds.txt"
echo "Updating builds.txt"
python ${GALAXY}/cron/parse_builds.py > ${GALAXY}/static/ucsc/new/builds.txt
python ${GALAXY}/cron/parse_builds.py > ${GALAXY}/tool-data/shared/ucsc/new/builds.txt
if [ $? -eq 0 ]
then
cp -uf ${GALAXY}/static/ucsc/new/builds.txt ${GALAXY}/static/ucsc/builds.txt
cp -uf ${GALAXY}/tool-data/shared/ucsc/new/builds.txt ${GALAXY}/tool-data/shared/ucsc/builds.txt
else
echo "Failed to update builds.txt" >&2
fi
# Try to build ucsc_build_sites.txt
echo "Updating ucsc_build_sites.txt"
python ${GALAXY}/cron/parse_builds_3_sites.py > ${GALAXY}/static/ucsc/new/ucsc_build_sites.txt
python ${GALAXY}/cron/parse_builds_3_sites.py > ${GALAXY}/tool-data/shared/ucsc/new/ucsc_build_sites.txt
if [ $? -eq 0 ]
then
cp -uf ${GALAXY}/static/ucsc/new/ucsc_build_sites.txt ${GALAXY}/static/ucsc/ucsc_build_sites.txt
cp -uf ${GALAXY}/tool-data/shared/ucsc/new/ucsc_build_sites.txt ${GALAXY}/tool-data/shared/ucsc/ucsc_build_sites.txt
else
echo "Failed to update builds.txt" >&2
fi
# Try to build chromInfo tables
echo "Building chromInfo tables."
python ${GALAXY}/cron/build_chrom_db.py ${GALAXY}/static/ucsc/chrom/new/ ${GALAXY}/static/ucsc/builds.txt
python ${GALAXY}/cron/build_chrom_db.py ${GALAXY}/tool-data/shared/ucsc/chrom/new/ ${GALAXY}/tool-data/shared/ucsc/builds.txt
if [ $? -eq 0 ]
then
cp -uf ${GALAXY}/static/ucsc/chrom/new/*.len ${GALAXY}/static/ucsc/chrom/
cp -uf ${GALAXY}/tool-data/shared/ucsc/chrom/new/*.len ${GALAXY}/tool-data/shared/ucsc/chrom/
else
echo "Failed to update chromInfo tables." >&2
fi
rm -rf ${GALAXY}/static/ucsc/new
rm -rf ${GALAXY}/static/ucsc/chrom/new
rm -rf ${GALAXY}/tool-data/shared/ucsc/new
rm -rf ${GALAXY}/tool-data/shared/ucsc/chrom/new
echo "Update complete."
#Perform Manual Additions here
echo "Adding Manual Builds."
python ${GALAXY}/cron/add_manual_builds.py ${GALAXY}/static/ucsc/manual_builds.txt ${GALAXY}/static/ucsc/builds.txt ${GALAXY}/static/ucsc/chrom/
python ${GALAXY}/cron/add_manual_builds.py ${GALAXY}/tool-data/shared/ucsc/manual_builds.txt ${GALAXY}/tool-data/shared/ucsc/builds.txt ${GALAXY}/tool-data/shared/ucsc/chrom/
if [ $? -eq 0 ]
then
echo "Manual addition was successful."
+11 -8
View File
@@ -242,7 +242,10 @@ def get_gbrowse_sites_by_build(build):
def read_dbnames(filename):
""" Read build names from file """
db_names = []
class DBNames( list ):
default_value = "?"
default_name = "unspecified (?)"
db_names = DBNames()
try:
ucsc_builds = {}
man_builds = [] #assume these are integers
@@ -276,15 +279,15 @@ def read_dbnames(filename):
ucsc_builds[db_base].sort()
ucsc_builds[db_base].reverse()
ucsc_builds[db_base] = [(build, name) for build_rev, build, name in ucsc_builds[db_base]]
db_names = db_names + ucsc_builds[db_base]
if len(db_names)>1 and len(man_builds)>0: db_names.append(('?', '----- Additional Species Are Below -----'))
db_names = DBNames( db_names + ucsc_builds[db_base] )
if len( db_names ) > 1 and len( man_builds ) > 0: db_names.append( ( db_names.default_value, '----- Additional Species Are Below -----' ) )
man_builds.sort()
man_builds = [(build, name) for name, build in man_builds]
db_names = db_names + man_builds
db_names = DBNames( db_names + man_builds )
except Exception, e:
print "ERROR: Unable to read builds file:", e
if len(db_names)<1:
db_names = [('?', 'unspecified (?)')]
db_names = DBNames( [( db_names.default_value, db_names.default_name )] )
return db_names
def read_build_sites(filename):
@@ -306,9 +309,9 @@ def read_build_sites(filename):
return build_sites
galaxy_root_path = os.path.join(__path__[0], "..","..","..")
dbnames = read_dbnames(os.path.join(galaxy_root_path,"static","ucsc","builds.txt")) #this list is used in edit attributes and the upload tool
ucsc_build_sites = read_build_sites(os.path.join(galaxy_root_path,"static","ucsc","ucsc_build_sites.txt")) #this list is used in history.tmpl
gbrowse_build_sites = read_build_sites(os.path.join(galaxy_root_path,"static","gbrowse","gbrowse_build_sites.txt")) #this list is used in history.tmpl
dbnames = read_dbnames( os.path.join( galaxy_root_path, "tool-data", "shared", "ucsc", "builds.txt" ) ) #this list is used in edit attributes and the upload tool
ucsc_build_sites = read_build_sites( os.path.join( galaxy_root_path, "tool-data", "shared", "ucsc", "ucsc_build_sites.txt" ) ) #this list is used in history.tmpl
gbrowse_build_sites = read_build_sites( os.path.join( galaxy_root_path, "tool-data", "shared", "gbrowse", "gbrowse_build_sites.txt" ) ) #this list is used in history.tmpl
if __name__ == '__main__':
import doctest, sys
+1 -1
View File
@@ -187,7 +187,7 @@ class UniverseWebTransaction( base.DefaultWebTransaction ):
initialize a new history with the first dbkey in util.dbnames which is currently
? unspecified (?)
"""
history.genome_build = util.dbnames[0][0]
history.genome_build = util.dbnames.default_value
if history.user_id is None and self.user is not None:
history.user_id = self.user.id
if self.galaxy_session_is_valid():
@@ -1,6 +1,6 @@
<tool id="Annotation_Profiler_0" name="Profile Annotations" Version="1.0.0">
<description>for a set of genomic intervals</description>
<command interpreter="python2.4">annotation_profiler_for_interval.py -i $input1 -c $input1_chromCol -s $input1_startCol -e $input1_endCol -o $out_file1 $keep_empty -p /depot/data2/galaxy/annotation_profiler/$dbkey $summary -b 3 -t $table_names</command>
<command interpreter="python2.4">annotation_profiler_for_interval.py -i $input1 -c $input1_chromCol -s $input1_startCol -e $input1_endCol -o $out_file1 $keep_empty -p /depot/data2/galaxy/annotation_profiler/$dbkey $summary -l ${GALAXY_DATA_INDEX_DIR}/shared/ucsc/chrom/${dbkey}.len -b 3 -t $table_names</command>
<inputs>
<param format="interval" name="input1" type="data" label="Choose Intervals">
<validator type="dataset_metadata_in_file" filename="annotation_profiler_valid_builds.txt" metadata_name="dbkey" metadata_column="0" message="Profiling is not currently available for this species."/>
@@ -111,8 +111,9 @@ class CachedCoverageReader:
yield tablename, coverage, regions, index
class TableCoverageSummary:
def __init__( self, coverage_reader ):
def __init__( self, coverage_reader, chrom_lengths ):
self.coverage_reader = coverage_reader
self.chrom_lengths = chrom_lengths
self.chromosome_coverage = {}
self.total_interval_size = 0
self.total_interval_count = 0
@@ -126,10 +127,7 @@ class TableCoverageSummary:
self.total_interval_size += ( end - start )
self.total_interval_count += 1
if chrom not in self.chromosome_coverage:
#utilize lengths file here, if possible, if not use 250mb
#currently, no valid method to provide location of lengths file by framework:
#gops_complement has it hard coded as dbfile = fileinput.FileInput( "static/ucsc/chrom/"+db+".len" )
self.chromosome_coverage[chrom] = bx.bitset.BitSet( 250000000 )
self.chromosome_coverage[chrom] = bx.bitset.BitSet( self.chrom_lengths.get( chrom ) )
self.chromosome_coverage[chrom].set_range( start, end - start )
for table_name, coverage, regions in self.coverage_reader.iter_table_coverage_regions_by_region( chrom, start, end ):
if table_name not in self.table_coverage:
@@ -207,19 +205,34 @@ def profile_per_interval( interval_filename, chrom_col, start_col, end_col, out_
out.write( "%s\t%s\t%s\t%s\n" % ( "\t".join( region.fields ), table_name, coverage, region_count ) )
out.close()
def profile_summary( interval_filename, chrom_col, start_col, end_col, out_filename, keep_empty, coverage_reader ):
def profile_summary( interval_filename, chrom_col, start_col, end_col, out_filename, keep_empty, coverage_reader, chrom_lengths ):
out = open( out_filename, 'wb' )
table_coverage_summary = TableCoverageSummary( coverage_reader )
table_coverage_summary = TableCoverageSummary( coverage_reader, chrom_lengths )
for region in bx.intervals.io.NiceReaderWrapper( open( interval_filename, 'rb' ), chrom_col = chrom_col, start_col = start_col, end_col = end_col, fix_strand = True, return_header = False, return_comments = False ):
table_coverage_summary.add_region( region.chrom, region.start, region.end )
out.write( "#tableName\ttableChromosomeCoverage\ttableChromosomeCount\ttableRegionCoverage\ttableRegionCount\tallIntervalCount\tallIntervalSize\tallCoverage\tallTableRegionsOverlaped\tallIntervalsOverlapingTable\tnrIntervalCount\tnrIntervalSize\tnrCoverage\tnrTableRegionsOverlaped\tnrIntervalsOverlapingTable\n" )#\tstatistic\n" )
out.write( "#tableName\ttableChromosomeCoverage\ttableChromosomeCount\ttableRegionCoverage\ttableRegionCount\tallIntervalCount\tallIntervalSize\tallCoverage\tallTableRegionsOverlaped\tallIntervalsOverlapingTable\tnrIntervalCount\tnrIntervalSize\tnrCoverage\tnrTableRegionsOverlaped\tnrIntervalsOverlapingTable\n" )
for table_name, table_chromosome_size, table_chromosome_count, table_region_coverage, table_region_count, total_interval_count, total_interval_size, total_coverage, table_regions_overlaped_count, interval_region_overlap_count, nr_interval_count, nr_region_size, nr_coverage, nr_table_regions_overlaped_count, nr_interval_table_overlap_count in table_coverage_summary.iter_table_coverage():
if keep_empty or total_coverage:
#only output tables that have atleast 1 base covered unless empty are requested
out.write( "%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\n" % ( table_name, table_chromosome_size, table_chromosome_count, table_region_coverage, table_region_count, total_interval_count, total_interval_size, total_coverage, table_regions_overlaped_count, interval_region_overlap_count, nr_interval_count, nr_region_size, nr_coverage, nr_table_regions_overlaped_count, nr_interval_table_overlap_count ) )
out.close()
class ChromosomeLengths:
def __init__( self, filename ):
self.chroms = {}
try:
for line in open( filename ):
try:
fields = line.strip().split( "\t" )
self.chroms[fields[0]] = int( fields[1] )
except:
continue
except:
pass
def get( self, name ):
return self.chroms.get( name, bx.bitset.MAX )
def __main__():
parser = optparse.OptionParser()
parser.add_option(
@@ -259,6 +272,12 @@ def __main__():
type='str',default='/depot/data2/galaxy/annotation_profiler/hg18',
help='Path to profiled data for this organism'
)
parser.add_option(
'-l','--lengths',
dest='lengths',
type='str',default='test-data/shared/ucsc/hg18.len',
help='Path to chromosome lengths data for this organism'
)
parser.add_option(
'-t','--table_names',
dest='table_names',
@@ -292,7 +311,7 @@ def __main__():
coverage_reader = CachedCoverageReader( options.path, buffer = options.buffer, table_names = table_names )
if options.summary:
profile_summary( options.interval_filename, options.chrom_col - 1, options.start_col - 1, options.end_col -1, options.out_filename, options.keep_empty, coverage_reader )
profile_summary( options.interval_filename, options.chrom_col - 1, options.start_col - 1, options.end_col -1, options.out_filename, options.keep_empty, coverage_reader, ChromosomeLengths( options.lengths ) )
else:
profile_per_interval( options.interval_filename, options.chrom_col - 1, options.start_col - 1, options.end_col -1, options.out_filename, options.keep_empty, coverage_reader )
+1 -1
View File
@@ -1,6 +1,6 @@
<tool id="axt_to_lav_1" name="AXT to LAV">
<description>Converts an AXT formated file to LAV format</description>
<command interpreter="python">axt_to_lav.py ${GALAXY_DATA_INDEX_DIR}/$dbkey_1/seq/%s.nib:$dbkey_1:./static/ucsc/chrom/${dbkey_1}.len ${GALAXY_DATA_INDEX_DIR}/$dbkey_2/seq/%s.nib:$dbkey_2:./static/ucsc/chrom/${dbkey_2}.len $align_input $lav_file $seq_file1 $seq_file2</command>
<command interpreter="python">axt_to_lav.py /depot/data2/galaxy/$dbkey_1/seq/%s.nib:$dbkey_1:${GALAXY_DATA_INDEX_DIR}/shared/ucsc/chrom/${dbkey_1}.len /depot/data2/galaxy/$dbkey_2/seq/%s.nib:$dbkey_2:${GALAXY_DATA_INDEX_DIR}/shared/ucsc/chrom/${dbkey_2}.len $align_input $lav_file $seq_file1 $seq_file2</command>
<inputs>
<param name="align_input" type="data" format="axt" label="Alignment File" optional="False"/>
<param name="dbkey_1" type="genomebuild" label="Genome"/>
+1 -1
View File
@@ -1,6 +1,6 @@
<tool id="gops_complement_1" name="Complement">
<description>intervals of a query</description>
<command interpreter="python">gops_complement.py $input1 $output -1 $input1_chromCol,$input1_startCol,$input1_endCol,$input1_strandCol -d $dbkey $allchroms</command>
<command interpreter="python">gops_complement.py $input1 $output -1 $input1_chromCol,$input1_startCol,$input1_endCol,$input1_strandCol -l ${GALAXY_DATA_INDEX_DIR}/shared/ucsc/chrom/${dbkey}.len $allchroms</command>
<inputs>
<param format="interval" name="input1" type="data">
<label>Complement regions of</label>
+3 -3
View File
@@ -4,7 +4,7 @@ Complement regions.
usage: %prog in_file out_file
-1, --cols1=N,N,N,N: Columns for chrom, start, end, strand in file
-d, --db=N: Database name (for determining chromosome lengths)
-l, --lengths=N: Filename of .len file for species (chromosome lengths)
-a, --all: Complement all chromosomes (Genome-wide complement)
"""
from galaxy import eggs
@@ -29,7 +29,7 @@ def main():
options, args = doc_optparse.parse( __doc__ )
try:
chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols1 )
db = options.db
lengths = options.lengths
if options.all: allchroms = True
in_fname, out_fname = args
except:
@@ -46,7 +46,7 @@ def main():
chroms = list()
# dbfile is used to determine the length of each chromosome. The lengths
# are added to the lens dict and passed copmlement operation code in bx.
dbfile = fileinput.FileInput( "static/ucsc/chrom/"+db+".len" )
dbfile = fileinput.FileInput( lengths )
if dbfile:
if not allchroms: