diff --git a/.hgignore b/.hgignore
index 72ceb2ad345..bc932e9b9ba 100644
--- a/.hgignore
+++ b/.hgignore
@@ -50,7 +50,9 @@ tool-data/gd.species.txt
tool-data/shared/igv/igv_build_sites.txt
tool-data/shared/rviewer/rviewer_build_sites.txt
tool-data/shared/ucsc/builds.txt
+tool-data/shared/ensembl/builds.txt
tool-data/*.loc
+tool-data/genome/*
# Test output
run_functional_tests.html
@@ -72,4 +74,5 @@ static/june_2007_style/blue/base_sprites.less
*.orig
.DS_Store
*.rej
-*~
\ No newline at end of file
+*~
+
diff --git a/cron/get_ensembl.py b/cron/get_ensembl.py
new file mode 100644
index 00000000000..decb984f773
--- /dev/null
+++ b/cron/get_ensembl.py
@@ -0,0 +1,22 @@
+from galaxy import eggs
+import pkg_resources
+pkg_resources.require("SQLAlchemy >= 0.4")
+pkg_resources.require("MySQL_python")
+from sqlalchemy import *
+
+
+engine = create_engine( 'mysql://anonymous@ensembldb.ensembl.org:5306', pool_recycle=3600 )
+conn = engine.connect()
+dbs = conn.execute( "SHOW DATABASES LIKE 'ensembl_website_%%'" )
+builds = {}
+lines = []
+for res in dbs:
+ dbname = res[0]
+ release = dbname.split('_')[-1]
+ genomes = conn.execute( "SELECT RS.assembly_code, S.name, S.common_name, %s FROM ensembl_website_%s.release_species RS LEFT JOIN ensembl_website_%s.species S on RS.species_id = S.species_id" % ( release, release, release ) )
+ for genome in genomes:
+ builds[genome[0]] = dict( release=genome[3], species='%s (%s/%s)' % ( genome[1], genome[2], genome[0] ) )
+for build in builds.items():
+ lines.append( '\t'.join( [ build[0], '%d' % build[1]['release'], build[1]['species'] ] ) )
+
+print '\n'.join( lines )
\ No newline at end of file
diff --git a/cron/parse_publicbuilds.py b/cron/parse_publicbuilds.py
new file mode 100644
index 00000000000..bb37f32d538
--- /dev/null
+++ b/cron/parse_publicbuilds.py
@@ -0,0 +1,57 @@
+#!/usr/bin/env python
+
+"""
+Connects to the URL specified and outputs builds available at that
+DSN in tabular format. USCS Test gateway is used as default.
+build description
+"""
+
+import sys
+import urllib
+if sys.version_info[:2] >= ( 2, 5 ):
+ import xml.etree.ElementTree as ElementTree
+else:
+ from galaxy import eggs
+ import pkg_resources; pkg_resources.require( "elementtree" )
+ from elementtree import ElementTree
+
+URL = "http://genome.cse.ucsc.edu/cgi-bin/das/dsn"
+
+def getbuilds(url):
+ try:
+ page = urllib.urlopen(URL)
+ except:
+ print "#Unable to open " + URL
+ print "?\tunspecified (?)"
+ sys.exit(1)
+
+ text = page.read()
+ try:
+ tree = ElementTree.fromstring(text)
+ except:
+ print "#Invalid xml passed back from " + URL
+ print "?\tunspecified (?)"
+ sys.exit(1)
+
+ print "#Harvested from http://genome.cse.ucsc.edu/cgi-bin/das/dsn"
+ print "?\tunspecified (?)"
+ for dsn in tree:
+ build = dsn.find("SOURCE").attrib['id']
+ description = dsn.find("DESCRIPTION").text.replace(" - Genome at UCSC","").replace(" Genome at UCSC","")
+
+ fields = description.split(" ")
+ temp = fields[0]
+ for i in range(len(fields)-1):
+ if temp == fields[i+1]:
+ fields.pop(i+1)
+ else:
+ temp = fields[i+1]
+ description = " ".join(fields)
+ yield [build,description]
+
+if __name__ == "__main__":
+ if len(sys.argv) > 1:
+ URL = sys.argv[1]
+ for build in getbuilds(URL):
+ print build[0]+"\t"+build[1]+" ("+build[0]+")"
+
diff --git a/cron/updateensembl.sh.sample b/cron/updateensembl.sh.sample
new file mode 100644
index 00000000000..612a11ca7b3
--- /dev/null
+++ b/cron/updateensembl.sh.sample
@@ -0,0 +1,42 @@
+#!/bin/sh
+#
+# Script to update Ensembl shared data tables. The idea is to update, but if
+# the update fails, not replace current data/tables with error
+# messages.
+
+# Edit this line to refer to galaxy's path:
+GALAXY=/path/to/galaxy
+PYTHONPATH=${GALAXY}/lib
+export PYTHONPATH
+
+# setup directories
+echo "Creating required directories."
+DIRS="
+${GALAXY}/tool-data/shared/ensembl
+${GALAXY}/tool-data/shared/ensembl/new
+"
+for dir in $DIRS; do
+ if [ ! -d $dir ]; then
+ echo "Creating $dir"
+ mkdir $dir
+ else
+ echo "$dir already exists, continuing."
+ fi
+done
+
+date
+echo "Updating Ensembl shared data tables."
+
+# Try to build "builds.txt"
+echo "Updating builds.txt"
+python ${GALAXY}/cron/get_ensembl.py > ${GALAXY}/tool-data/shared/ensembl/new/builds.txt
+if [ $? -eq 0 ]
+then
+ diff ${GALAXY}/tool-data/shared/ensembl/new/builds.txt ${GALAXY}/tool-data/shared/ensembl/builds.txt > /dev/null 2>&1
+ if [ $? -ne 0 ]
+ then
+ cp -f ${GALAXY}/tool-data/shared/ensembl/new/builds.txt ${GALAXY}/tool-data/shared/ensembl/builds.txt
+ fi
+else
+ echo "Failed to update builds.txt" >&2
+fi
diff --git a/cron/updateucsc.sh.sample b/cron/updateucsc.sh.sample
index 4ebaf825821..5d157d86a35 100644
--- a/cron/updateucsc.sh.sample
+++ b/cron/updateucsc.sh.sample
@@ -28,6 +28,20 @@ done
date
echo "Updating UCSC shared data tables."
+# Try to build "publicbuilds.txt"
+echo "Updating publicbuilds.txt"
+python ${GALAXY}/cron/parse_publicbuilds.py > ${GALAXY}/tool-data/shared/ucsc/new/publicbuilds.txt
+if [ $? -eq 0 ]
+then
+ diff ${GALAXY}/tool-data/shared/ucsc/new/publicbuilds.txt ${GALAXY}/tool-data/shared/ucsc/publicbuilds.txt > /dev/null 2>&1
+ if [ $? -ne 0 ]
+ then
+ cp -f ${GALAXY}/tool-data/shared/ucsc/new/publicbuilds.txt ${GALAXY}/tool-data/shared/ucsc/publicbuilds.txt
+ fi
+else
+ echo "Failed to update publicbuilds.txt" >&2
+fi
+
# Try to build "builds.txt"
echo "Updating builds.txt"
python ${GALAXY}/cron/parse_builds.py > ${GALAXY}/tool-data/shared/ucsc/new/builds.txt
diff --git a/lib/galaxy/jobs/deferred/genome_index.py b/lib/galaxy/jobs/deferred/genome_index.py
new file mode 100644
index 00000000000..7572df5b109
--- /dev/null
+++ b/lib/galaxy/jobs/deferred/genome_index.py
@@ -0,0 +1,43 @@
+"""
+Module for managing genome transfer jobs.
+"""
+from __future__ import with_statement
+
+import logging, shutil, gzip, bz2, zipfile, tempfile, tarfile, sys, os
+
+from galaxy import eggs
+from sqlalchemy import and_
+from data_transfer import *
+
+log = logging.getLogger( __name__ )
+
+__all__ = [ 'GenomeIndexPlugin' ]
+
+class GenomeIndexPlugin( DataTransfer ):
+
+ def __init__( self, app ):
+ super( GenomeIndexPlugin, self ).__init__( app )
+ self.app = app
+ self.tool = app.toolbox.tools_by_id['__GENOME_INDEX__']
+ self.sa_session = app.model.context.current
+
+ def create_job( self, trans, path, indexes, dbkey, intname ):
+ params = dict( user=trans.user.id, path=path, indexes=indexes, dbkey=dbkey, intname=intname )
+ deferred = trans.app.model.DeferredJob( state = self.app.model.DeferredJob.states.NEW, plugin = 'GenomeIndexPlugin', params = params )
+ self.sa_session.add( deferred )
+ self.sa_session.flush()
+ log.debug( 'Job created, id %d' % deferred.id )
+ return deferred.id
+
+ def check_job( self, job ):
+ log.debug( 'Job check' )
+ return 'ready'
+
+ def run_job( self, job ):
+ incoming = dict( path=os.path.abspath( job.params[ 'path' ] ), indexer=job.params[ 'indexes' ][0], user=job.params[ 'user' ] )
+ indexjob = self.tool.execute( self, set_output_hid=False, history=None, incoming=incoming, transfer=None, deferred=job )
+ job.params[ 'indexjob' ] = indexjob[0].id
+ job.state = self.app.model.DeferredJob.states.RUNNING
+ self.sa_session.add( job )
+ self.sa_session.flush()
+ return self.app.model.DeferredJob.states.RUNNING
diff --git a/lib/galaxy/jobs/deferred/genome_transfer.py b/lib/galaxy/jobs/deferred/genome_transfer.py
index cb011cbb2a8..2fb9a4719cf 100644
--- a/lib/galaxy/jobs/deferred/genome_transfer.py
+++ b/lib/galaxy/jobs/deferred/genome_transfer.py
@@ -78,10 +78,11 @@ class GenomeTransferPlugin( DataTransfer ):
def get_job_status( self, jobid ):
job = self.sa_session.query( self.app.model.DeferredJob ).get( int( jobid ) )
- if not hasattr( job, 'transfer_job' ):
- job.transfer_job = self.sa_session.query( self.app.model.TransferJob ).get( int( job.params[ 'transfer_job_id' ] ) )
- else:
- self.sa_session.refresh( job.transfer_job )
+ if 'transfer_job_id' in job.params:
+ if not hasattr( job, 'transfer_job' ):
+ job.transfer_job = self.sa_session.query( self.app.model.TransferJob ).get( int( job.params[ 'transfer_job_id' ] ) )
+ else:
+ self.sa_session.refresh( job.transfer_job )
return job
def run_job( self, job ):
@@ -139,7 +140,6 @@ class GenomeTransferPlugin( DataTransfer ):
if not chunk:
break
os.write( fd, chunk )
- os.write( fd, '\n' )
os.close( fd )
compressed.close()
elif data_type == 'bzip':
@@ -154,7 +154,6 @@ class GenomeTransferPlugin( DataTransfer ):
if not chunk:
break
os.write( fd, chunk )
- os.write( fd, '\n' )
os.close( fd )
compressed.close()
elif data_type == 'zip':
@@ -177,7 +176,6 @@ class GenomeTransferPlugin( DataTransfer ):
if not chunk:
break
os.write( fd, chunk )
- os.write( fd, '\n' )
zipped_file.close()
else:
try:
@@ -223,8 +221,8 @@ class GenomeTransferPlugin( DataTransfer ):
else:
job.state = self.app.model.DeferredJob.states.OK
self.sa_session.add( job )
- return self.app.model.DeferredJob.states.OK
self.sa_session.flush()
+ return self.app.model.DeferredJob.states.OK
def _check_compress( self, filepath ):
retval = ''
diff --git a/lib/galaxy/jobs/deferred/liftover_transfer.py b/lib/galaxy/jobs/deferred/liftover_transfer.py
index 6bdd3df735f..ecc8638be63 100644
--- a/lib/galaxy/jobs/deferred/liftover_transfer.py
+++ b/lib/galaxy/jobs/deferred/liftover_transfer.py
@@ -40,7 +40,7 @@ class LiftOverTransferPlugin( DataTransfer ):
deferred = trans.app.model.DeferredJob( state = self.app.model.DeferredJob.states.NEW, plugin = 'LiftOverTransferPlugin', params = params )
self.sa_session.add( deferred )
self.sa_session.flush()
- return deferred.id
+ return job.id
def check_job( self, job ):
if job.params['type'] == 'init_transfer':
@@ -98,7 +98,9 @@ class LiftOverTransferPlugin( DataTransfer ):
transfer = job.transfer_job
if params[ 'type' ] == 'extract_transfer':
CHUNK_SIZE = 2**20
- destpath = os.path.join( self.app.config.get( 'genome_data_path', 'tool-data/genome' ), job.params[ 'dbkey' ], 'liftOver' )
+ destpath = os.path.join( self.app.config.get( 'genome_data_path', 'tool-data/genome' ), source, 'liftOver' )
+ if not os.path.exists( destpath ):
+ os.makedirs( destpath )
destfile = job.params[ 'destfile' ]
destfilepath = os.path.join( destpath, destfile )
tmpprefix = '%s_%s_download_unzip_' % ( job.params['dbkey'], job.params[ 'transfer_job_id' ] )
diff --git a/lib/galaxy/tools/actions/index_genome.py b/lib/galaxy/tools/actions/index_genome.py
index 08e7f698965..e59d31e1e0c 100644
--- a/lib/galaxy/tools/actions/index_genome.py
+++ b/lib/galaxy/tools/actions/index_genome.py
@@ -21,7 +21,9 @@ class GenomeIndexToolAction( ToolAction ):
job.tool_id = tool.id
job.user_id = incoming['user']
start_job_state = job.state # should be job.states.NEW
- job.state = job.states.WAITING # we need to set job state to something other than NEW, or else when tracking jobs in db it will be picked up before we have added input / output parameters
+ job.state = job.states.WAITING # we need to set job state to something other than NEW,
+ # or else when tracking jobs in db it will be picked up
+ # before we have added input / output parameters
trans.sa_session.add( job )
# Create dataset that will serve as archive.
diff --git a/lib/galaxy/tools/genome_index/__init__.py b/lib/galaxy/tools/genome_index/__init__.py
index e13bdfdbed8..6318325f226 100644
--- a/lib/galaxy/tools/genome_index/__init__.py
+++ b/lib/galaxy/tools/genome_index/__init__.py
@@ -13,13 +13,12 @@ log = logging.getLogger(__name__)
def load_genome_index_tools( toolbox ):
""" Adds tools for indexing genomes via the main job runner. """
- # Use same process as that used in load_external_metadata_tool; see that
- # method for why create tool description files on the fly.
+ # Create XML for loading the tool.
tool_xml_text = """
- $__GENOME_INDEX_COMMAND__ $output_file $output_file.files_path $__app__.config.rsync_url
+ $__GENOME_INDEX_COMMAND__ $output_file $output_file.files_path $__app__.config.rsync_url "$__app__.config.tool_data_path"
@@ -29,7 +28,7 @@ def load_genome_index_tools( toolbox ):
"""
- # Load export tool.
+ # Load index tool.
tmp_name = tempfile.NamedTemporaryFile()
tmp_name.write( tool_xml_text )
tmp_name.flush()
@@ -166,6 +165,10 @@ class GenomeIndexToolWrapper( object ):
self._check_link( fasta, target )
for line in location:
self._add_line( line[ 'file' ], line[ 'line' ] )
+ deferred.state = app.model.DeferredJob.states.OK
+ sa_session.add( deferred )
+ sa_session.flush()
+
def _check_link( self, targetfile, symlink ):
target = os.path.relpath( targetfile, os.path.dirname( symlink ) )
diff --git a/lib/galaxy/tools/genome_index/index_genome.py b/lib/galaxy/tools/genome_index/index_genome.py
index d191b161e13..26aa918fe9f 100644
--- a/lib/galaxy/tools/genome_index/index_genome.py
+++ b/lib/galaxy/tools/genome_index/index_genome.py
@@ -10,7 +10,8 @@ from __future__ import with_statement
import optparse, sys, os, tempfile, time, subprocess, shlex, json, tarfile, shutil
class ManagedIndexer():
- def __init__( self, output_file, infile, workingdir, rsync_url ):
+ def __init__( self, output_file, infile, workingdir, rsync_url, tooldata ):
+ self.tooldatapath = os.path.abspath( tooldata )
self.workingdir = os.path.abspath( workingdir )
self.outfile = open( os.path.abspath( output_file ), 'w' )
self.basedir = os.path.split( self.workingdir )[0]
@@ -44,11 +45,12 @@ class ManagedIndexer():
with WithChDir( self.workingdir ):
self._log( 'Running indexer %s.' % indexer )
result = getattr( self, self.indexers[ indexer ] )()
- if result is None:
- self._log( 'Error running indexer %s.' % indexer )
+ if result in [ None, False ]:
+ self._log( 'Error running indexer %s, %s' % ( indexer, result ) )
self._flush_files()
return True
else:
+ self._log( self.locations )
self._log( 'Indexer %s completed successfully.' % indexer )
self._flush_files()
@@ -93,6 +95,7 @@ class ManagedIndexer():
os.remove( self.fafile )
return self._bwa_cs()
else:
+ self._log( 'BWA (base) exited with code %s' % result )
return False
def _bwa_cs( self ):
@@ -109,6 +112,7 @@ class ManagedIndexer():
self.locations[ 'cs' ].append( self.fafile )
os.remove( self.fafile )
else:
+ self._log( 'BWA (color) exited with code %s' % result )
return False
else:
self.locations[ 'cs' ].append( self.fafile )
@@ -136,6 +140,7 @@ class ManagedIndexer():
os.remove( self.fafile )
return self._bowtie_cs()
else:
+ self._log( 'Bowtie (base) exited with code %s' % result )
return False
def _bowtie_cs( self ):
@@ -149,6 +154,7 @@ class ManagedIndexer():
if result == 0:
self.locations[ 'cs' ].append( self.genome )
else:
+ self._log( 'Bowtie (color) exited with code %s' % result )
return False
os.remove( os.path.join( indexdir, self.fafile ) )
else:
@@ -174,6 +180,7 @@ class ManagedIndexer():
os.remove( self.fafile )
return True
else:
+ self._log( 'Bowtie2 exited with code %s' % result )
return False
def _twobit( self ):
@@ -193,6 +200,7 @@ class ManagedIndexer():
os.remove( self.fafile )
return True
else:
+ self._log( 'faToTwoBit exited with code %s' % result )
return False
def _perm( self ):
@@ -208,12 +216,15 @@ class ManagedIndexer():
command = shlex.split("PerM %s %s --readFormat fastq --seed %s -m -s %s" % (self.fafile, read_length, seed, index))
result = subprocess.call( command )
if result != 0:
+ self._log( 'PerM (base) exited with code %s' % result )
return False
self.locations[ 'nt' ].append( [ key, desc, index ] )
os.remove( self.fafile )
return self._perm_cs()
def _perm_cs( self ):
+ genome = self.genome
+ read_length = 50
if not os.path.exists( 'cs' ):
os.makedirs( 'cs' )
with WithChDir( 'cs' ):
@@ -223,12 +234,13 @@ class ManagedIndexer():
desc = '%s: seed=%s, read length=%s' % (genome, seed, read_length)
index = "%s_color_%s_%s.index" % (genome, seed, read_length)
if not os.path.exists( index ):
- command = shlex.split("PerM %s %s --readFormat csfastq --seed %s -m -s %s" % (local_ref, read_length, seed, index))
+ command = shlex.split("PerM %s %s --readFormat csfastq --seed %s -m -s %s" % (self.fafile, read_length, seed, index))
result = subprocess.call( command, stderr=self.logfile, stdout=self.logfile )
if result != 0:
+ self._log( 'PerM (color) exited with code %s' % result )
return False
self.locations[ 'cs' ].append( [ key, desc, index ] )
- os.remove( local_ref )
+ os.remove( self.fafile )
temptar = tarfile.open( 'cs.tar', 'w' )
temptar.add( 'cs' )
temptar.close()
@@ -241,17 +253,19 @@ class ManagedIndexer():
self.locations[ 'nt' ].append( self.fafile )
return True
local_ref = self.fafile
- srma = 'tool-data/shared/jars/srma.jar'
+ srma = os.path.abspath( os.path.join( self.tooldatapath, 'shared/jars/picard/CreateSequenceDictionary.jar' ) )
genome = os.path.splitext( self.fafile )[0]
self._check_link()
if not os.path.exists( '%s.fai' % self.fafile ) and not os.path.exists( '%s.fai' % self.genome ):
command = shlex.split( 'samtools faidx %s' % self.fafile )
subprocess.call( command, stderr=self.logfile )
- command = shlex.split( "java -cp %s net.sf.picard.sam.CreateSequenceDictionary R=%s O=%s/%s.dict URI=%s" \
- % ( srma, local_ref, os.curdir, genome, local_ref ) )
+ command = shlex.split( "java -jar %s R=%s O=%s.dict URI=%s" \
+ % ( srma, local_ref, genome, local_ref ) )
if not os.path.exists( '%s.dict' % self.genome ):
result = subprocess.call( command, stderr=self.logfile, stdout=self.logfile )
+ self._log( ' '.join( command ) )
if result != 0:
+ self._log( 'Picard exited with code %s' % result )
return False
self.locations[ 'nt' ].append( self.fafile )
os.remove( self.fafile )
@@ -260,17 +274,20 @@ class ManagedIndexer():
def _sam( self ):
local_ref = self.fafile
local_file = os.path.splitext( self.fafile )[ 0 ]
+ print 'Trying rsync'
result = self._do_rsync( '/sam_index/' )
if result == 0 and ( os.path.exists( '%s.fai' % self.fafile ) or os.path.exists( '%s.fai' % self.genome ) ):
- self.locations[ 'nt' ].append( local_ref )
+ self.locations[ 'nt' ].append( '%s.fai' % local_ref )
return True
self._check_link()
+ print 'Trying indexer'
command = shlex.split("samtools faidx %s" % local_ref)
- result = subprocess.call( command, stderr=self.logfile )
+ result = subprocess.call( command, stderr=self.logfile, stdout=self.logfile )
if result != 0:
+ self._log( 'SAM exited with code %s' % result )
return False
else:
- self.locations[ 'nt' ].append( local_ref )
+ self.locations[ 'nt' ].append( '%s.fai' % local_ref )
os.remove( local_ref )
return True
@@ -288,9 +305,9 @@ if __name__ == "__main__":
# Parse command line.
parser = optparse.OptionParser()
(options, args) = parser.parse_args()
- indexer, infile, outfile, working_dir, rsync_url = args
+ indexer, infile, outfile, working_dir, rsync_url, tooldata = args
# Create archive.
- idxobj = ManagedIndexer( outfile, infile, working_dir, rsync_url )
+ idxobj = ManagedIndexer( outfile, infile, working_dir, rsync_url, tooldata )
idxobj.run_indexer( indexer )
\ No newline at end of file
diff --git a/lib/galaxy/util/__init__.py b/lib/galaxy/util/__init__.py
index 1b77cfb9414..3c6e738f827 100644
--- a/lib/galaxy/util/__init__.py
+++ b/lib/galaxy/util/__init__.py
@@ -407,6 +407,22 @@ def read_dbnames(filename):
db_names = DBNames( [( db_names.default_value, db_names.default_name )] )
return db_names
+def read_ensembl( filename, ucsc ):
+ """ Read Ensembl build names from file """
+ ucsc_builds = []
+ for build in ucsc:
+ ucsc_builds.append( build[0] )
+ ensembl_builds = list()
+ try:
+ for line in open( filename ):
+ if line[0:1] in [ '#', '\t' ]: continue
+ fields = line.replace("\r","").replace("\n","").split("\t")
+ if fields[0] in ucsc_builds: continue
+ ensembl_builds.append( dict( dbkey=fields[0], release=fields[1], name=fields[2].replace( '_', ' ' ) ) )
+ except Exception, e:
+ print "ERROR: Unable to read builds file:", e
+ return ensembl_builds
+
def read_build_sites( filename, check_builds=True ):
""" read db names to ucsc mappings from file, this file should probably be merged with the one above """
build_sites = []
@@ -634,11 +650,15 @@ def send_mail( frm, to, subject, body, config ):
s.quit()
galaxy_root_path = os.path.join(__path__[0], "..","..","..")
+
# The dbnames list is used in edit attributes and the upload tool
dbnames = read_dbnames( os.path.join( galaxy_root_path, "tool-data", "shared", "ucsc", "builds.txt" ) )
+ucsc_names = read_dbnames( os.path.join( galaxy_root_path, "tool-data", "shared", "ucsc", "publicbuilds.txt" ) )
+ensembl_names = read_ensembl( os.path.join( galaxy_root_path, "tool-data", "shared", "ensembl", "builds.txt" ), ucsc_names )
ucsc_build_sites = read_build_sites( os.path.join( galaxy_root_path, "tool-data", "shared", "ucsc", "ucsc_build_sites.txt" ) )
gbrowse_build_sites = read_build_sites( os.path.join( galaxy_root_path, "tool-data", "shared", "gbrowse", "gbrowse_build_sites.txt" ) )
genetrack_sites = read_build_sites( os.path.join( galaxy_root_path, "tool-data", "shared", "genetrack", "genetrack_sites.txt" ), check_builds=False )
+dlnames = dict(ucsc=ucsc_names, ensembl=ensembl_names)
def galaxy_directory():
return os.path.abspath(galaxy_root_path)
diff --git a/lib/galaxy/web/controllers/data_admin.py b/lib/galaxy/web/controllers/data_admin.py
index 66beb20e364..a7a49e04316 100644
--- a/lib/galaxy/web/controllers/data_admin.py
+++ b/lib/galaxy/web/controllers/data_admin.py
@@ -26,15 +26,67 @@ class DataAdmin( BaseUIController ):
error='panel-error-message',
queued='state-color-waiting'
)
-
+
@web.expose
@web.require_admin
def manage_data( self, trans, **kwd ):
+ genomes = dict()
if trans.app.config.get_bool( 'enable_beta_job_managers', False ) == False:
return trans.fill_template( '/admin/data_admin/betajob.mako' )
- dbkeys = trans.db_builds
- return trans.fill_template( '/admin/data_admin/data_form.mako', dbkeys=dbkeys )
+ for line in trans.app.tool_data_tables.data_tables[ 'all_fasta' ].data:
+ indexers = dict( bowtie_indexes='Generate', bowtie2_indexes='Generate', bwa_indexes='Generate', perm_base_indexes='Generate', srma_indexes='Generate', sam_fa_indexes='Generate' )
+ dbkey = line[0]
+ name = line[2]
+ indexers[ 'name' ] = name
+ indexers[ 'fapath' ] = line[3]
+ genomes[ dbkey ] = indexers
+ for table in [ 'bowtie_indexes', 'bowtie2_indexes', 'bwa_indexes', 'srma_indexes' ]:
+ for line in trans.app.tool_data_tables.data_tables[ table ].data:
+ dbkey = line[0]
+ genomes[ dbkey ][ table ] = 'Generated'
+ for line in trans.app.tool_data_tables.data_tables[ 'sam_fa_indexes' ].data:
+ genomes[ line[1] ][ 'sam_fa_indexes' ] = 'Generated'
+ for line in trans.app.tool_data_tables.data_tables[ 'perm_base_indexes' ].data:
+ genomes[ line[1].split(':')[0] ][ 'perm_base_indexes' ] = 'Generated'
+ jobgrid = []
+ sa_session = trans.app.model.context.current
+ jobs = sa_session.query( model.GenomeIndexToolData ).order_by( model.GenomeIndexToolData.created_time.desc() ).filter_by( user_id=trans.get_user().id ).group_by( model.GenomeIndexToolData.deferred ).limit( 20 ).all()
+ prevjobid = 0
+ for job in jobs:
+ if prevjobid == job.deferred.id:
+ continue
+ prevjobid = job.deferred.id
+ state = job.deferred.state
+ params = job.deferred.params
+ if job.transfer is not None:
+ jobtype = 'download'
+ else:
+ jobtype = 'index'
+ indexers = ', '.join( params['indexes'] )
+ jobgrid.append( dict( jobtype=jobtype, indexers=indexers, rowclass=state, deferred=job.deferred.id, state=state, intname=job.deferred.params[ 'intname' ], dbkey=job.deferred.params[ 'dbkey' ] ) )
+ return trans.fill_template( '/admin/data_admin/local_data.mako', jobgrid=jobgrid, genomes=genomes )
+
+ @web.expose
+ @web.require_admin
+ def add_genome( self, trans, **kwd ):
+ if trans.app.config.get_bool( 'enable_beta_job_managers', False ) == False:
+ return trans.fill_template( '/admin/data_admin/betajob.mako' )
+ dbkeys = trans.ucsc_builds
+ ensemblkeys = trans.ensembl_builds
+ return trans.fill_template( '/admin/data_admin/data_form.mako', dbkeys=dbkeys, ensembls=ensemblkeys )
+ @web.expose
+ @web.require_admin
+ def index_build( self, trans, **kwd ):
+ """Index a previously downloaded genome."""
+ params = util.Params( kwd )
+ path = os.path.abspath( params.get( 'path', None ) )
+ indexes = [ params.get( 'indexes', None ) ]
+ dbkey = params.get( 'dbkey', None )
+ intname = params.get( 'longname', None )
+ indexjob = trans.app.job_manager.deferred_job_queue.plugins['GenomeIndexPlugin'].create_job( trans, path, indexes, dbkey, intname )
+ return indexjob
+
@web.expose
@web.require_admin
def download_build( self, trans, **kwd ):
@@ -57,21 +109,21 @@ class DataAdmin( BaseUIController ):
protocol = 'http'
if source == 'NCBI':
- dbkey = params.get('dbkey', '')[0]
+ dbkey = params.get('ncbi_dbkey', '')[0]
url = 'http://togows.dbcls.jp/entry/ncbi-nucleotide/%s.fasta' % dbkey
elif source == 'Broad':
- dbkey = params.get('dbkey', '')[0]
+ dbkey = params.get('broad_dbkey', '')[0]
url = 'ftp://ftp.broadinstitute.org/pub/seq/references/%s.fasta' % dbkey
elif source == 'UCSC':
longname = None
- for build in trans.db_builds:
- if dbkey[1] == build[0]:
+ for build in trans.ucsc_builds:
+ if dbkey == build[0]:
dbkey = build[0]
longname = build[1]
break
assert dbkey is not '?', 'That build was not found'
ftp = ftplib.FTP('hgdownload.cse.ucsc.edu')
- ftp.login('anonymous', 'user@example.com')
+ ftp.login('anonymous', trans.get_user().email)
checker = []
liftover = []
newlift = []
@@ -81,10 +133,12 @@ class DataAdmin( BaseUIController ):
fname = chain.split( '/' )[-1]
target = fname.replace( '.over.chain.gz', '' ).split( 'To' )[1]
target = target[0].lower() + target[1:]
- newlift.append( [ chain, dbkey, target ] )
+ if not os.path.exists( os.path.join( trans.app.config.get( 'genome_data_path', 'tool-data/genome' ), dbkey, 'liftOver', fname ) ):
+ newlift.append( [ chain, dbkey, target ] )
current = dbkey[0].upper() + dbkey[1:]
targetfile = '%sTo%s.over.chain.gz' % ( target, current )
- newlift.append( [ '/goldenPath/%s/liftOver/%s' % ( target, targetfile ), target, dbkey ] )
+ if not os.path.exists( os.path.join( trans.app.config.get( 'genome_data_path', 'tool-data/genome' ), target, 'liftOver', targetfile ) ):
+ newlift.append( [ '/goldenPath/%s/liftOver/%s' % ( target, targetfile ), target, dbkey ] )
except:
newlift = None
pass
@@ -103,36 +157,35 @@ class DataAdmin( BaseUIController ):
status = u'error'
return trans.fill_template( '/admin/data_admin/data_form.mako',
message=message,
- status=status )
+ status=status,
+ ensembls=trans.ensembl_builds,
+ dbkeys=trans.ucsc_builds )
elif source == 'Ensembl':
- section = params.get('ensembl_section', '')
- release1 = params.get('release_number', '')
- organism = params.get('organism', '')
- name = params.get('name', '')
- longname = organism
- dbkey = name
- release2 = params.get('release2', '')
- release2 = ".%s" % release2 if release2 else ""
- if section == 'standard':
- url = 'ftp://ftp.ensembl.org/pub/release-%s/fasta/%s/dna/%s.%s%s.dna.toplevel.fa.gz' % \
- (release1, organism.lower(), organism, name, release2)
- else:
- url = 'ftp://ftp.ensemblgenomes.org/pub/%s/release-%s/fasta/%s/dna/%s.%s%s.dna.toplevel.fa.gz' % \
- (section, release1, organism.lower(), organism, name, release2)
- elif source == 'local':
- url = 'http://127.0.0.1/%s.tar.gz' % dbkey
+ dbkey = params.get( 'ensembl_dbkey', None )
+ assert dbkey is not '?', 'That build was not found'
+ for build in trans.ensembl_builds:
+ if build[ 'dbkey' ] == dbkey:
+ dbkey = build[ 'dbkey' ]
+ release = build[ 'release' ]
+ pathname = '_'.join( build[ 'name' ].split(' ')[0:2] )
+ longname = build[ 'name' ].replace('_', ' ')
+ break
+ url = 'ftp://ftp.ensembl.org/pub/release-%s/fasta/%s/dna/%s.%s.%s.dna.toplevel.fa.gz' % ( release, pathname.lower(), pathname, dbkey, release )
+ log.debug( build )
+ log.debug( url )
else:
- raise ValueError
+ raise ValueError, 'Somehow an invalid data source was specified.'
params = dict( protocol='http', name=dbkey, datatype='fasta', url=url, user=trans.user.id )
jobid = trans.app.job_manager.deferred_job_queue.plugins['GenomeTransferPlugin'].create_job( trans, url, dbkey, longname, indexers )
chainjob = []
if newlift is not None:
for chain in newlift:
- liftover_url = u'ftp://hgdownload.cse.ucsc.edu%s' % chain[0]
+ liftover_url = u'ftp://hgdownload.cse.ucsc.edu%s' % chain[0]
from_genome = chain[1]
to_genome = chain[2]
destfile = liftover_url.split('/')[-1].replace('.gz', '')
- chainjob.append( trans.app.job_manager.deferred_job_queue.plugins['LiftOverTransferPlugin'].create_job( trans, liftover_url, dbkey, from_genome, to_genome, destfile, jobid ) )
+ lochain = trans.app.job_manager.deferred_job_queue.plugins['LiftOverTransferPlugin'].create_job( trans, liftover_url, dbkey, from_genome, to_genome, destfile, jobid )
+ chainjob.append( lochain )
job = trans.app.job_manager.deferred_job_queue.plugins['GenomeTransferPlugin'].get_job_status( jobid )
job.params['liftover'] = chainjob
trans.app.model.context.current.add( job )
@@ -146,9 +199,13 @@ class DataAdmin( BaseUIController ):
def monitor_status( self, trans, **kwd ):
params = util.Params( kwd )
jobid = params.get( 'job', '' )
+ gname = params.get( 'intname', '' )
+ deferred = trans.app.model.context.current.query( model.DeferredJob ).filter_by( id=jobid ).first()
+ gname = deferred.params[ 'intname' ]
+ indexers = ', '.join( deferred.params[ 'indexes' ] )
jobs = self._get_jobs( jobid, trans )
jsonjobs = json.dumps( jobs )
- return trans.fill_template( '/admin/data_admin/download_status.mako', mainjob=jobid, jobs=jobs, jsonjobs=jsonjobs )
+ return trans.fill_template( '/admin/data_admin/download_status.mako', name=gname, indexers=indexers, mainjob=jobid, jobs=jobs, jsonjobs=jsonjobs )
@web.expose
@web.require_admin
@@ -160,16 +217,6 @@ class DataAdmin( BaseUIController ):
jobs = self._get_jobs( jobid, trans )
return trans.fill_template( '/admin/data_admin/ajax_status.mako', json=json.dumps( jobs ) )
- @web.expose
- @web.require_admin
- def job_status( self, trans, **kwd ):
- params = util.Params( kwd )
- jobid = params.get( 'jobid', None )
- jobtype = params.get( 'jobtype', None )
- fillvals = None
- fillvals = self._get_job( jobid, jobtype, trans )
- return trans.fill_template( '/admin/data_admin/ajax_status.mako', json=json.dumps( fillvals ) )
-
def _get_job( self, jobid, jobtype, trans ):
sa = trans.app.model.context.current
if jobtype == 'liftover':
@@ -191,12 +238,12 @@ class DataAdmin( BaseUIController ):
job = trans.app.job_manager.deferred_job_queue.plugins['GenomeTransferPlugin'].get_job_status( jobid )
sa_session = trans.app.model.context.current
jobs.append( self._get_job( job.id, 'deferred', trans ) )
- jobs.append( self._get_job( job.transfer_job.id, 'transfer', trans ) )
- idxjobs = sa_session.query( model.GenomeIndexToolData ).filter_by( deferred_job_id=job.id, transfer_job_id=job.transfer_job.id ).all()
- if job.params.has_key( 'liftover' ):
- for jobid in job.params[ 'liftover' ]:
- jobs.append( self._get_job( jobid, 'liftover', trans ) )
- for idxjob in idxjobs:
- #print idxjob
- jobs.append( self._get_job( idxjob.job_id, 'index', trans ) )
+ if hasattr( job, 'transfer_job' ): # This is a transfer job, check for indexers
+ jobs.append( self._get_job( job.transfer_job.id, 'transfer', trans ) )
+ idxjobs = sa_session.query( model.GenomeIndexToolData ).filter_by( deferred_job_id=job.id, transfer_job_id=job.transfer_job.id ).all()
+ if job.params.has_key( 'liftover' ) and job.params[ 'liftover' ] is not None:
+ for jobid in job.params[ 'liftover' ]:
+ jobs.append( self._get_job( jobid, 'liftover', trans ) )
+ for idxjob in idxjobs:
+ jobs.append( self._get_job( idxjob.job_id, 'index', trans ) )
return jobs
diff --git a/lib/galaxy/web/framework/__init__.py b/lib/galaxy/web/framework/__init__.py
index b22ba924ac6..9417dbdd52b 100644
--- a/lib/galaxy/web/framework/__init__.py
+++ b/lib/galaxy/web/framework/__init__.py
@@ -799,6 +799,14 @@ class GalaxyWebTransaction( base.DefaultWebTransaction ):
dbnames.extend( util.dbnames )
return dbnames
+ @property
+ def ucsc_builds( self ):
+ return util.dlnames['ucsc']
+
+ @property
+ def ensembl_builds( self ):
+ return util.dlnames['ensembl']
+
def db_dataset_for( self, dbkey ):
"""
Returns the db_file dataset associated/needed by `dataset`, or `None`.
@@ -957,6 +965,14 @@ class GalaxyWebAPITransaction( GalaxyWebTransaction ):
dbnames.append((key, "%s (%s) [Custom]" % (chrom_dict['name'], key) ))
dbnames.extend( util.dbnames )
return dbnames
+
+ @property
+ def ucsc_builds( self ):
+ return util.dlnames['ucsc']
+
+ @property
+ def ensembl_builds( self ):
+ return util.dlnames['ensembl']
class GalaxyWebUITransaction( GalaxyWebTransaction ):
def __init__( self, environ, app, webapp, session_cookie ):
diff --git a/templates/admin/data_admin/data_form.mako b/templates/admin/data_admin/data_form.mako
index 2092747a358..c57aa200b99 100644
--- a/templates/admin/data_admin/data_form.mako
+++ b/templates/admin/data_admin/data_form.mako
@@ -62,7 +62,7 @@
Parameters
-