From df82f3e163e152bc9034e693453d70f72de49f33 Mon Sep 17 00:00:00 2001 From: Daniel Blankenberg Date: Tue, 3 Mar 2009 12:54:06 -0500 Subject: [PATCH] Add the ability for metadata to be set in an external process, this requires set_metadata_externally = True in universe_wsgi.ini. Currently this is supported in the local runner (long running set_meta() can now be canceled the same as with jobs) and in the non-staging PBS runner. There is a slight change to how new Metadata FileParameters are to be created in datatype.set_meta(), see the MAF datatype. TODO: Prevent editing of metadata on a running job when metadata is to be set externally. Have datasets which were copied before job completion use the metadata from the original output dataset; currently .set_meta() is called locally on these. --- lib/galaxy/config.py | 1 + lib/galaxy/datatypes/metadata.py | 197 +++++++++++++++++++++++++++-- lib/galaxy/datatypes/sequence.py | 6 +- lib/galaxy/jobs/__init__.py | 35 ++++- lib/galaxy/jobs/runners/local.py | 26 +++- lib/galaxy/jobs/runners/pbs.py | 7 +- lib/galaxy/model/__init__.py | 15 +++ lib/galaxy/model/mapping.py | 19 ++- lib/galaxy/tools/actions/upload.py | 3 +- lib/galaxy/util/__init__.py | 37 +++++- lib/galaxy/web/controllers/root.py | 2 +- scripts/set_metadata.py | 38 ++++++ set_metadata.sh | 4 + universe_wsgi.ini.sample | 3 + 14 files changed, 365 insertions(+), 28 deletions(-) create mode 100644 scripts/set_metadata.py create mode 100755 set_metadata.sh diff --git a/lib/galaxy/config.py b/lib/galaxy/config.py index a3cdb4b4b4a..4e481f7f9ce 100644 --- a/lib/galaxy/config.py +++ b/lib/galaxy/config.py @@ -37,6 +37,7 @@ class Configuration( object ): self.tool_config = resolve_path( kwargs.get( 'tool_config_file', 'tool_conf.xml' ), self.root ) self.tool_secret = kwargs.get( "tool_secret", "" ) self.id_secret = kwargs.get( "id_secret", "USING THE DEFAULT IS NOT SECURE!" ) + self.set_metadata_externally = string_as_bool( kwargs.get( "set_metadata_externally", "False" ) ) self.use_remote_user = string_as_bool( kwargs.get( "use_remote_user", "False" ) ) self.remote_user_maildomain = kwargs.get( "remote_user_maildomain", None ) self.require_login = string_as_bool( kwargs.get( "require_login", "False" ) ) diff --git a/lib/galaxy/datatypes/metadata.py b/lib/galaxy/datatypes/metadata.py index f0e6e9d6958..8865cbfcda7 100644 --- a/lib/galaxy/datatypes/metadata.py +++ b/lib/galaxy/datatypes/metadata.py @@ -1,6 +1,6 @@ -import sys, logging, copy, shutil, weakref +import sys, logging, copy, shutil, weakref, cPickle, tempfile, os -from galaxy.util import string_as_bool +from galaxy.util import string_as_bool, relpath from galaxy.util.odict import odict from galaxy.web import form_builder import galaxy.model @@ -9,6 +9,8 @@ log = logging.getLogger( __name__ ) STATEMENTS = "__galaxy_statements__" #this is the name of the property in a Datatype class where new metadata spec element Statements are stored +DATABASE_CONNECTION_AVAILABLE = True #When False, certain metadata parameter types (see FileParameter) will behave differently + class Statement( object ): """ This class inserts its target into a list in the surrounding @@ -28,7 +30,7 @@ class Statement( object ): statement.target( element, *args, **kwargs ) #statement.target is MetadataElementSpec, element is a Datatype class -class MetadataCollection: +class MetadataCollection( object ): """ MetadataCollection is not a collection at all, but rather a proxy to the real metadata which is stored as a Dictionary. This class @@ -90,6 +92,26 @@ class MetadataCollection: if key in self.spec: rval[key] = self.spec[key].param.make_copy( value, target_context=self, source_context=to_copy ) return rval + def from_pickled_dict( self, filename ): + dataset = self.parent + log.debug( 'loading metadata from file for: %s %s' % ( dataset.__class__.__name__, dataset.id ) ) + pickled_dict = cPickle.load( open( filename ) ) + for name, spec in self.spec.items(): + if name in pickled_dict: + dataset._metadata[ name ] = spec.param.from_pickle_value( pickled_dict[ name ], dataset ) + elif name in dataset._metadata: + #if the metadata value is not found in our externally set metadata but it has a value in the 'old' + #metadata associated with our dataset, we'll delete it from our dataset's metadata dict + del dataset._metadata[ name ] + def to_pickled_dict( self, filename ): + meta_dict = {} + dataset_meta_dict = self.parent._metadata + for name, spec in self.spec.items(): + if name in dataset_meta_dict: + meta_dict[ name ] = spec.param.to_pickle_value( dataset_meta_dict[ name ] ) + cPickle.dump( meta_dict, open( filename, 'wb+' ) ) + def __getstate__( self ): + return None #cannot pickle a weakref item (self._parent), when data._metadata_collection is None, it will be recreated on demand class MetadataSpecCollection( odict ): """ @@ -168,6 +190,17 @@ class MetadataParameter( object ): """ return value + def from_pickle_value( self, value, parent ): + """ + Turns a value read from a pickled dict into its value to be pushed directly into the metadata dict. + """ + return value + def to_pickle_value( self, value ): + """ + Turns a value read from a metadata into its value to be pushed directly into the pickled dict. + """ + return value + class MetadataElementSpec( object ): """ Defines a metadata element and adds it to the metadata_spec (which @@ -326,14 +359,19 @@ class FileParameter( MetadataParameter ): return "
No display available for Metadata Files
" def wrap( self, value ): - if isinstance( value, galaxy.model.MetadataFile ): + if isinstance( value, galaxy.model.MetadataFile ) or isinstance( value, MetadataTempFile ): return value - try: - return galaxy.model.MetadataFile.get( value ) - except: - #value was not a valid id - return None - + if DATABASE_CONNECTION_AVAILABLE: + try: + return galaxy.model.MetadataFile.get( value ) + except: + #value was not a valid id + return None + elif value is not None: + mf = galaxy.model.MetadataFile() + mf.id = value #we assume this is a valid id, since we cannot check it + return mf + return None def make_copy( self, value, target_context = None, source_context = None ): value = self.wrap( value ) if value: @@ -342,9 +380,146 @@ class FileParameter( MetadataParameter ): shutil.copy( value.file_name, new_value.file_name ) return self.unwrap( new_value ) return None - + @classmethod def marshal( cls, value ): if isinstance( value, galaxy.model.MetadataFile ): value = value.id return value + + def from_pickle_value( self, value, parent ): + """ + Turns a value read from a pickled dict into its value to be pushed directly into the metadata dict. + """ + if isinstance( value, MetadataTempFile ): + mf = self.new_file( dataset = parent, **value.kwd ) + shutil.move( value.file_name, mf.file_name ) + value = mf.id + return value + def to_pickle_value( self, value ): + """ + Turns a value read from a metadata into its value to be pushed directly into the pickled dict. + """ + if isinstance( value, galaxy.model.MetadataFile ): + value = value.id + return value + + def new_file( self, dataset = None, **kwd ): + if DATABASE_CONNECTION_AVAILABLE: + mf = galaxy.model.MetadataFile( name = self.spec.name, dataset = dataset, **kwd ) + mf.flush() #flush to assign id + return mf + else: + #we need to make a tmp file that is accessable to the head node, + #we will be copying its contents into the MetadataFile objects filename after restoring from pickle + #we do not include 'dataset' in the kwds passed, as from_pickle_value() will handle this for us + return MetadataTempFile( **kwd ) + +#This class is used when a database file connection is not available +class MetadataTempFile( object ): + tmp_dir = 'database/tmp' #this should be overwritten as necessary in calling scripts + def __init__( self, **kwd ): + self.kwd = kwd + self._filename = None + @property + def file_name( self ): + if self._filename is None: + #we need to create a tmp file, accessable across all nodes/heads, save the name, and return it + self._filename = relpath( tempfile.NamedTemporaryFile( dir = self.tmp_dir, prefix = "metadata_temp_file_" ).name ) + open( self._filename, 'wb+' ) #create an empty file, so it can't be reused using tempfile + return self._filename + @classmethod + def cleanup_from_pickled_dict_filename( cls, filename ): + try: + for key, value in cPickle.load( open( filename ) ).items(): + if isinstance( value, cls ) and os.path.exists( value.file_name ): + log.debug( 'Cleaning up abandoned MetadataTempFile file: %s' % value.file_name ) + os.unlink( value.file_name ) + except Exception, e: + log.debug( 'Failed to cleanup MetadataTempFile temp files from %s: %s' % ( filename, e ) ) + +#Class with methods allowing set_meta() to be called externally to the Galaxy head +class JobExternalOutputMetadataWrapper( object ): + #this class allows access to external metadata filenames for all outputs associated with a job + def __init__( self, job ): + self.job_id = job.id + def get_output_filenames_by_dataset( self, dataset ): + if isinstance( dataset, galaxy.model.HistoryDatasetAssociation ): + return galaxy.model.JobExternalOutputMetadata.filter_by( job_id = self.job_id, history_dataset_association_id = dataset.id ).first() #there should only be one or None + elif isinstance( dataset, galaxy.model.LibraryDatasetDatasetAssociation ): + return galaxy.model.JobExternalOutputMetadata.filter_by( job_id = self.job_id, library_dataset_dataset_association_id = dataset.id ).first() #there should only be one or None + return None + def get_dataset_metadata_key( self, dataset ): + return "%s_%d" % ( dataset.__class__.__name__, dataset.id ) #set meta can be called on library items and history items, need to make different keys for them, since ids can overlap + def setup_external_metadata( self, datasets, exec_dir = None, tmp_dir = None, dataset_files_path = None, kwds = {} ): + #fill in metadata_files_dict and return the command with args required to set metadata + def __metadata_files_list_to_cmd_line( metadata_files ): + return "%s,%s,%s,%s" % ( metadata_files.filename_in, metadata_files.filename_kwds, metadata_files.filename_out, metadata_files.filename_results_code ) + if not isinstance( datasets, list ): + datasets = [ datasets ] + if exec_dir is None: + exec_dir = os.path.abspath( os.getcwd() ) + if tmp_dir is None: + tmp_dir = MetadataTempFile.tmp_dir + if dataset_files_path is None: + dataset_files_path = galaxy.model.Dataset.file_path + metadata_files_list = [] + for dataset in datasets: + key = self.get_dataset_metadata_key( dataset ) + #future note: + #wonkiness in job execution causes build command line to be called more than once + #when setting metadata externally, via 'auto-detect' button in edit attributes, etc., + #we don't want to overwrite (losing the ability to cleanup) our existing dataset keys and files, + #so we will only populate the dictionary once + metadata_files = self.get_output_filenames_by_dataset( dataset ) + if not metadata_files: + metadata_files = galaxy.model.JobExternalOutputMetadata( dataset = dataset) + metadata_files.job_id = self.job_id + #we are using tempfile to create unique filenames, tempfile always returns an absolute path + #we will use pathnames relative to the galaxy root, to accommodate instances where the galaxy root + #is located differently, i.e. on a cluster node with a different filesystem structure + + #file to store existing dataset + metadata_files.filename_in = relpath( tempfile.NamedTemporaryFile( dir = tmp_dir, prefix = "metadata_in_%s_" % key ).name ) + cPickle.dump( dataset, open( metadata_files.filename_in, 'wb+' ) ) + #file to store metadata results of set_meta() + metadata_files.filename_out = relpath( tempfile.NamedTemporaryFile( dir = tmp_dir, prefix = "metadata_out_%s_" % key ).name ) + open( metadata_files.filename_out, 'wb+' ) # create the file on disk, so it cannot be reused by tempfile (unlikely, but possible) + #file to store a 'return code' indicating the results of the set_meta() call + #results code is like (True/False - if setting metadata was successful/failed , exception or string of reason of success/failure ) + metadata_files.filename_results_code = relpath( tempfile.NamedTemporaryFile( dir = tmp_dir, prefix = "metadata_out_%s_" % key ).name ) + cPickle.dump( ( False, 'External set_meta() not called' ), open( metadata_files.filename_results_code, 'wb+' ) ) # create the file on disk, so it cannot be reused by tempfile (unlikely, but possible) + #file to store kwds passed to set_meta() + metadata_files.filename_kwds = relpath( tempfile.NamedTemporaryFile( dir = tmp_dir, prefix = "metadata_kwds_%s_" % key ).name ) + cPickle.dump( kwds, open( metadata_files.filename_kwds, 'wb+' ) ) + metadata_files.flush() + metadata_files_list.append( metadata_files ) + #return command required to build + return "%s %s %s %s" % ( os.path.join( exec_dir, 'set_metadata.sh' ), dataset_files_path, tmp_dir, " ".join( map( __metadata_files_list_to_cmd_line, metadata_files_list ) ) ) + + def external_metadata_set_successfully( self, dataset ): + metadata_files = self.get_output_filenames_by_dataset( dataset ) + if not metadata_files: + return False # this file doesn't exist + rval, rstring = cPickle.load( open( metadata_files.filename_results_code ) ) + if not rval: + log.debug( 'setting metadata externally failed for %s %s: %s' % ( dataset.__class__.__name__, dataset.id, rstring ) ) + return rval + + def cleanup_external_metadata( self ): + log.debug( 'Cleaning up external metadata files' ) + for metadata_files in galaxy.model.Job.get( self.job_id ).external_output_metadata: + #we need to confirm that any MetadataTempFile files were removed, if not we need to remove them + #can occur if the job was stopped before completion, but a MetadataTempFile is used in the set_meta + MetadataTempFile.cleanup_from_pickled_dict_filename( metadata_files.filename_out ) + dataset_key = self.get_dataset_metadata_key( metadata_files.dataset ) + for key, fname in [ ( 'filename_in', metadata_files.filename_in ), ( 'filename_out', metadata_files.filename_out ), ( 'filename_results_code', metadata_files.filename_results_code ), ( 'filename_kwds', metadata_files.filename_kwds ) ]: + try: + os.remove( fname ) + except Exception, e: + log.debug( 'Failed to cleanup external metadata file (%s) for %s: %s' % ( key, dataset_key, e ) ) + def set_job_runner_external_pid( self, pid ): + for metadata_files in galaxy.model.Job.get( self.job_id ).external_output_metadata: + metadata_files.job_runner_external_pid = pid + metadata_files.flush() + diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 10d54cffa03..d8d9f076af9 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -254,8 +254,7 @@ class Maf( Alignment ): tmp_file.write( "%s\t%s\n" % ( spec, "\t".join( chroms ) ) ) if not chrom_file: - chrom_file = galaxy.model.MetadataFile( dataset = dataset, name = "species_chromosomes" ) - chrom_file.flush() + chrom_file = dataset.metadata.spec['species_chromosomes'].param.new_file( dataset = dataset ) tmp_file.seek( 0 ) open( chrom_file.file_name, 'wb' ).write( tmp_file.read() ) dataset.metadata.species_chromosomes = chrom_file @@ -263,8 +262,7 @@ class Maf( Alignment ): index_file = dataset.metadata.maf_index if not index_file: - index_file = galaxy.model.MetadataFile( dataset = dataset, name="maf_index" ) - index_file.flush() + index_file = dataset.metadata.spec['maf_index'].param.new_file( dataset = dataset ) indexes.write( open( index_file.file_name, 'w' ) ) dataset.metadata.maf_index = index_file diff --git a/lib/galaxy/jobs/__init__.py b/lib/galaxy/jobs/__init__.py index 792da30a250..91962e3c9ff 100644 --- a/lib/galaxy/jobs/__init__.py +++ b/lib/galaxy/jobs/__init__.py @@ -4,6 +4,7 @@ from galaxy import util, model from galaxy.model import mapping from galaxy.datatypes.tabular import * from galaxy.datatypes.interval import * +from galaxy.datatypes import metadata import pkg_resources pkg_resources.require( "PasteDeploy" ) @@ -306,6 +307,7 @@ class JobWrapper( object ): self.working_directory = \ os.path.join( self.app.config.job_working_directory, str( self.job_id ) ) self.output_paths = None + self.external_output_metadata = metadata.JobExternalOutputMetadataWrapper( job ) #wrapper holding the info required to restore and clean up from pickled files used for setting metadata externally def get_param_dict( self ): """ @@ -462,6 +464,7 @@ class JobWrapper( object ): self.fail( "Job %s's output dataset(s) could not be read" % job.id ) return for dataset_assoc in job.output_datasets: + #should this also be checking library associations? - can a library item be added from a history before the job has ended? - lets not allow this to occur for dataset in dataset_assoc.dataset.dataset.history_associations: #need to update all associated output hdas, i.e. history was shared with job running dataset.blurb = 'done' dataset.peek = 'no peek' @@ -470,13 +473,25 @@ class JobWrapper( object ): if stderr: dataset.blurb = "error" elif dataset.has_data(): - # Only set metadata values if they are missing... - dataset.set_meta( overwrite = False ) + #if a dataset was copied, it won't appear in our dictionary: + #either use the metadata from originating output dataset, or call set_meta on the copies + #it would be quicker to just copy the metadata from the originating output dataset, + #but somewhat trickier (need to recurse up the copied_from tree), for now we'll call set_meta() + if not self.external_output_metadata.external_metadata_set_successfully( dataset ): + # Only set metadata values if they are missing... + dataset.set_meta( overwrite = False ) + else: + #load metadata from file + #we need to no longer allow metadata to be edited while the job is still running, + #since if it is edited, the metadata changed on the running output will no longer match + #the metadata that was stored to disk for use via the external process, + #and the changes made by the user will be lost, without warning or notice + dataset.metadata.from_pickled_dict( self.external_output_metadata.get_output_filenames_by_dataset( dataset ).filename_out ) dataset.set_peek() else: dataset.blurb = "empty" dataset.flush() - if stderr: + if stderr: dataset_assoc.dataset.dataset.state = model.Dataset.states.ERROR else: dataset_assoc.dataset.dataset.state = model.Dataset.states.OK @@ -517,10 +532,12 @@ class JobWrapper( object ): def cleanup( self ): # remove temporary files try: - for fname in self.extra_filenames: + for fname in self.extra_filenames: os.remove( fname ) if self.working_directory is not None: shutil.rmtree( self.working_directory ) + if self.app.config.set_metadata_externally: + self.external_output_metadata.cleanup_external_metadata() except: log.exception( "Unable to cleanup job %d" % self.job_id ) @@ -573,7 +590,15 @@ class JobWrapper( object ): for outfile in [ str( o ) for o in output_paths ]: sizes.append( ( outfile, os.stat( outfile ).st_size ) ) return sizes - + def setup_external_metadata( self, exec_dir = None, tmp_dir = None, dataset_files_path = None, **kwds ): + if tmp_dir is None: + #this dir should should relative to the exec_dir + tmp_dir = self.app.config.new_file_path + if dataset_files_path is None: + dataset_files_path = self.app.model.Dataset.file_path + job = model.Job.get( self.job_id ) + return self.external_output_metadata.setup_external_metadata( [ output_dataset_assoc.dataset for output_dataset_assoc in job.output_datasets ], exec_dir = exec_dir, tmp_dir = tmp_dir, dataset_files_path = dataset_files_path, **kwds ) + class DefaultJobDispatcher( object ): def __init__( self, app ): self.app = app diff --git a/lib/galaxy/jobs/runners/local.py b/lib/galaxy/jobs/runners/local.py index a2d675a70b6..7d07a04a0da 100644 --- a/lib/galaxy/jobs/runners/local.py +++ b/lib/galaxy/jobs/runners/local.py @@ -99,6 +99,21 @@ class LocalJobRunner( object ): job_wrapper.fail( "failure running job", exception=True ) log.exception("failure running job %d" % job_wrapper.job_id) return + + #run the metadata setting script here + #this is terminatable when output dataset/job is deleted + #so that long running set_meta()s can be cancelled without having to reboot the server + if job_wrapper.get_state() not in [ model.Job.states.ERROR, model.Job.states.DELETED ] and self.app.config.set_metadata_externally: + external_metadata_script = job_wrapper.setup_external_metadata( kwds = { 'overwrite' : False } ) #we don't want to overwrite metadata that was copied over in init_meta(), as per established behavior + log.debug( 'executing external set_meta script for job %d: %s' % ( job_wrapper.job_id, external_metadata_script ) ) + external_metadata_proc = subprocess.Popen( args = external_metadata_script, + shell = True, + env = env, + preexec_fn = os.setpgrp ) + job_wrapper.external_output_metadata.set_job_runner_external_pid( external_metadata_proc.pid ) + external_metadata_proc.wait() + log.debug( 'execution of external set_meta finished for job %d' % job_wrapper.job_id ) + # Finish the job try: job_wrapper.finish( stdout, stderr ) @@ -131,12 +146,17 @@ class LocalJobRunner( object ): return False def stop_job( self, job ): - if job.job_runner_external_id is None: + #if our local job has JobExternalOutputMetadata associated, then our primary job has to have already finished + if job.external_output_metadata: + pid = job.external_output_metadata[0].job_runner_external_pid #every JobExternalOutputMetadata has a pid set, we just need to take from one of them + else: + pid = job.job_runner_external_id + if pid in [ None, '' ]: log.warning( "stop_job(): %s: no PID in database for job, unable to stop" % job.id ) return - pid = int( job.job_runner_external_id ) + pid = int( pid ) if not self.check_pid( pid ): - log.warning( "stop_job(): %s: PID %d was already dead or can't be signaled" %job.id ) + log.warning( "stop_job(): %s: PID %d was already dead or can't be signaled" % ( job.id, pid ) ) return for sig in [ 15, 9 ]: try: diff --git a/lib/galaxy/jobs/runners/pbs.py b/lib/galaxy/jobs/runners/pbs.py index 91c0044effc..257ca66b4de 100644 --- a/lib/galaxy/jobs/runners/pbs.py +++ b/lib/galaxy/jobs/runners/pbs.py @@ -27,6 +27,7 @@ if [ "$GALAXY_LIB" != "None" ]; then fi cd %s %s +%s """ pbs_symlink_template = """#!/bin/sh @@ -208,7 +209,11 @@ class PBSJobRunner( object ): if self.app.config.pbs_stage_path != '': script = pbs_symlink_template % (job_wrapper.galaxy_lib_dir, " ".join(job_wrapper.get_input_fnames() + job_wrapper.get_output_fnames()), self.app.config.pbs_stage_path, exec_dir, command_line) else: - script = pbs_template % (job_wrapper.galaxy_lib_dir, exec_dir, command_line) + if self.app.config.set_metadata_externally: + external_metadata_script = job_wrapper.setup_external_metadata( exec_dir = exec_dir, tmp_dir = self.app.config.new_file_path, dataset_files_path = self.app.model.Dataset.file_path, kwds = { 'overwrite' : False } ) #we don't want to overwrite metadata that was copied over in init_meta(), as per established behavior + else: + external_metadata_script = "" + script = pbs_template % ( job_wrapper.galaxy_lib_dir, exec_dir, command_line, external_metadata_script ) job_file = "%s/%s.sh" % (self.app.config.cluster_files_directory, job_wrapper.job_id) fh = file(job_file, "w") fh.write(script) diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 9078726e596..46fb08bf684 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -141,6 +141,21 @@ class JobToOutputDatasetAssociation( object ): self.name = name self.dataset = dataset +class JobExternalOutputMetadata( object ): + def __init__( self, job = None, dataset = None ): + self.job = job + if isinstance( dataset, galaxy.model.HistoryDatasetAssociation ): + self.history_dataset_association = dataset + elif isinstance( dataset, galaxy.model.LibraryDatasetDatasetAssociation ): + self.library_dataset_dataset_association = dataset + @property + def dataset( self ): + if self.history_dataset_association: + return self.history_dataset_association + elif self.library_dataset_dataset_association: + return self.library_dataset_dataset_association + return None + class Group( object ): def __init__( self, name = None ): self.name = name diff --git a/lib/galaxy/model/mapping.py b/lib/galaxy/model/mapping.py index f9f0f8004b3..b547adc49d9 100644 --- a/lib/galaxy/model/mapping.py +++ b/lib/galaxy/model/mapping.py @@ -402,6 +402,17 @@ JobToOutputDatasetAssociation.table = Table( "job_to_output_dataset", metadata, Column( "dataset_id", Integer, ForeignKey( "history_dataset_association.id" ), index=True ), Column( "name", String(255) ) ) +JobExternalOutputMetadata.table = Table( "job_external_output_metadata", metadata, + Column( "id", Integer, primary_key=True ), + Column( "job_id", Integer, ForeignKey( "job.id" ), index=True ), + Column( "history_dataset_association_id", Integer, ForeignKey( "history_dataset_association.id" ), index=True, nullable=True ), + Column( "library_dataset_dataset_association_id", Integer, ForeignKey( "library_dataset_dataset_association.id" ), index=True, nullable=True ), + Column( "filename_in", String( 255 ) ), + Column( "filename_out", String( 255 ) ), + Column( "filename_results_code", String( 255 ) ), + Column( "filename_kwds", String( 255 ) ), + Column( "job_runner_external_pid", String( 255 ) ) ) + Event.table = Table( "event", metadata, Column( "id", Integer, primary_key=True ), Column( "create_time", DateTime, default=now ), @@ -780,12 +791,18 @@ assign_mapper( context, JobToOutputDatasetAssociation, JobToOutputDatasetAssocia assign_mapper( context, JobParameter, JobParameter.table ) +assign_mapper( context, JobExternalOutputMetadata, JobExternalOutputMetadata.table, + properties=dict( job = relation( Job ), + history_dataset_association = relation( HistoryDatasetAssociation, lazy = False ), + library_dataset_dataset_association = relation( LibraryDatasetDatasetAssociation, lazy = False ) ) ) + assign_mapper( context, Job, Job.table, properties=dict( galaxy_session=relation( GalaxySession ), history=relation( History ), parameters=relation( JobParameter, lazy=False ), input_datasets=relation( JobToInputDatasetAssociation, lazy=False ), - output_datasets=relation( JobToOutputDatasetAssociation, lazy=False ) ) ) + output_datasets=relation( JobToOutputDatasetAssociation, lazy=False ), + external_output_metadata = relation( JobExternalOutputMetadata, lazy = False ) ) ) assign_mapper( context, Event, Event.table, properties=dict( history=relation( History ), diff --git a/lib/galaxy/tools/actions/upload.py b/lib/galaxy/tools/actions/upload.py index 56048488050..2c74a436b44 100644 --- a/lib/galaxy/tools/actions/upload.py +++ b/lib/galaxy/tools/actions/upload.py @@ -126,7 +126,8 @@ class UploadToolAction( object ): return self.upload_empty( trans, job, "Empty file error:", "you attempted to upload an empty file." ) elif len( data_list ) < 1: return self.upload_empty( trans, job, "No data error:", "either you pasted no data, the url you specified is invalid, or you have not specified a file." ) - hda = data_list[0] + #if we could make a 'real' job here, then metadata could be set before job.finish() is called + hda = data_list[0] #only our first hda is being added as input for the job, why? job.state = trans.app.model.Job.states.OK file_size_str = datatypes.data.nice_size( hda.dataset.file_size ) job.info = "%s, size: %s" % ( hda.info, file_size_str ) diff --git a/lib/galaxy/util/__init__.py b/lib/galaxy/util/__init__.py index 117d1012c21..21937620cf6 100644 --- a/lib/galaxy/util/__init__.py +++ b/lib/galaxy/util/__init__.py @@ -3,7 +3,7 @@ Utility functions used systemwide. """ import logging -import threading, sets, random, string, md5, re, binascii, pickle, time, datetime, math, re, os +import threading, sets, random, string, md5, re, binascii, pickle, time, datetime, math, re, os, sys import pkg_resources @@ -376,6 +376,41 @@ def read_build_sites(filename): print "ERROR: Unable to read builds for site file %s" %filename return build_sites +def relpath( path, start = None ): + """Return a relative version of a path""" + #modified from python 2.6.1 source code + + #version 2.6+ has it built in, we'll use the 'official' copy + if sys.version_info[:2] >= ( 2, 6 ): + if start is not None: + return os.path.relpath( path, start ) + return os.path.relpath( path ) + + #we need to initialize some local parameters + curdir = os.curdir + pardir = os.pardir + sep = os.sep + commonprefix = os.path.commonprefix + join = os.path.join + if start is None: + start = curdir + + #below is the unedited (but formated) relpath() from posixpath.py of 2.6.1 + #this will likely not function properly on non-posix systems, i.e. windows + if not path: + raise ValueError( "no path specified" ) + + start_list = os.path.abspath( start ).split( sep ) + path_list = os.path.abspath( path ).split( sep ) + + # Work out how much of the filepath is shared by start and path. + i = len( commonprefix( [ start_list, path_list ] ) ) + + rel_list = [ pardir ] * ( len( start_list )- i ) + path_list[ i: ] + if not rel_list: + return curdir + return join( *rel_list ) + galaxy_root_path = os.path.join(__path__[0], "..","..","..") dbnames = read_dbnames( os.path.join( galaxy_root_path, "tool-data", "shared", "ucsc", "builds.txt" ) ) #this list is used in edit attributes and the upload tool ucsc_build_sites = read_build_sites( os.path.join( galaxy_root_path, "tool-data", "shared", "ucsc", "ucsc_build_sites.txt" ) ) #this list is used in history.tmpl diff --git a/lib/galaxy/web/controllers/root.py b/lib/galaxy/web/controllers/root.py index bc2986a7442..780cc42fd39 100644 --- a/lib/galaxy/web/controllers/root.py +++ b/lib/galaxy/web/controllers/root.py @@ -250,7 +250,7 @@ class RootController( BaseController ): if name not in [ 'name', 'info', 'dbkey' ]: if spec.get( 'default' ): setattr( data.metadata, name, spec.unwrap( spec.get( 'default' ) ) ) - data.datatype.set_meta( data ) + data.set_meta() data.datatype.after_edit( data ) trans.app.model.flush() return trans.show_ok_message( "Attributes updated", refresh_frames=['history'] ) diff --git a/scripts/set_metadata.py b/scripts/set_metadata.py new file mode 100644 index 00000000000..7cd7ad2cc0b --- /dev/null +++ b/scripts/set_metadata.py @@ -0,0 +1,38 @@ +""" +Execute an external process to set_meta() on a provided list of pickled datasets. + +This should not be called directly! Use the set_metadata.sh script in Galaxy's +top level directly. + +""" + +import os, sys, cPickle +assert sys.version_info[:2] >= ( 2, 4 ) + +new_path = [ os.path.join( os.getcwd(), "lib" ) ] +new_path.extend( sys.path[1:] ) # remove scripts/ from the path +sys.path = new_path + +from galaxy import eggs +import pkg_resources +import galaxy.model.mapping #need to load this before we unpickle, in order to setup properties assigned by the mappers +galaxy.model.Job() #this looks REAL stupid, but it is REQUIRED in order for SA to insert parameters into the classes defined by the mappers --> it appears that instantiating ANY mapper'ed class would suffice here +galaxy.datatypes.metadata.DATABASE_CONNECTION_AVAILABLE = False #Let metadata know that there is no database connection, and to just assume object ids are valid + +def __main__(): + file_path = sys.argv.pop( 1 ) + tmp_dir = sys.argv.pop( 1 ) + galaxy.model.Dataset.file_path = file_path + galaxy.datatypes.metadata.MetadataTempFile.tmp_dir = tmp_dir + for pickled_filenames in sys.argv[1:]: + pickled_filename_in, pickled_filename_kwds, pickled_filename_out, pickled_filename_results_code = pickled_filenames.split( ',' ) + try: + data = cPickle.load( open( pickled_filename_in ) )#unpickle DatasetInstance + kwds = cPickle.load( open( pickled_filename_kwds ) )#unpickle kwds + data.datatype.set_meta( data, **kwds ) + data.metadata.to_pickled_dict( pickled_filename_out ) # write out results of set_meta + cPickle.dump( ( True, 'Metadata has been set successfully' ), open( pickled_filename_results_code, 'wb+' ) ) #setting metadata has suceeded + except Exception, e: + cPickle.dump( ( False, e ), open( pickled_filename_results_code, 'wb+' ) ) #setting metadata has failed somehow + +__main__() \ No newline at end of file diff --git a/set_metadata.sh b/set_metadata.sh new file mode 100755 index 00000000000..962c799e75c --- /dev/null +++ b/set_metadata.sh @@ -0,0 +1,4 @@ +#!/bin/sh + +cd `dirname $0` +python -ES ./scripts/set_metadata.py $@ diff --git a/universe_wsgi.ini.sample b/universe_wsgi.ini.sample index ab89ce4c554..d0739961b04 100644 --- a/universe_wsgi.ini.sample +++ b/universe_wsgi.ini.sample @@ -40,6 +40,9 @@ tool_data_path = tool-data datatype_converters_config_file = datatype_converters_conf.xml datatype_converters_path = %(here)s/lib/galaxy/datatypes/converters +# Metadata +set_metadata_externally = False + # Session support (beaker) use_beaker_session = True session_type = file