From d8ef35b97a98d1d3a023dc5f5b8eefa139fa9187 Mon Sep 17 00:00:00 2001 From: John Duddy Date: Tue, 24 May 2011 15:15:03 -0700 Subject: [PATCH 001/362] Added new "multi" parallelism option for tools --- lib/galaxy/datatypes/data.py | 102 +++++++++++++++++++- lib/galaxy/datatypes/sequence.py | 89 +++++++++++++++++ lib/galaxy/jobs/__init__.py | 56 +++++++---- lib/galaxy/jobs/runners/tasks.py | 73 ++++++-------- lib/galaxy/jobs/splitters/basic.py | 104 ++++---------------- lib/galaxy/jobs/splitters/multi.py | 149 +++++++++++++++++++++++++++++ lib/galaxy/model/__init__.py | 11 +++ lib/galaxy/tools/__init__.py | 17 +++- 8 files changed, 449 insertions(+), 152 deletions(-) create mode 100644 lib/galaxy/jobs/splitters/multi.py diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index e37744b69d3..9e0cf4c0b41 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -1,4 +1,4 @@ -import logging, os, sys, time, tempfile +import logging, os, sys, time, tempfile, gzip from galaxy import util from galaxy.util.odict import odict from galaxy.util.bunch import Bunch @@ -351,6 +351,32 @@ class Data( object ): @property def has_resolution(self): return False + + + + def merge( split_files, output_file): + """ + Export files are usually compressed, but it doesn't have to be so. In the case that they are, use + zcat to cat the files and gzip -c to recompress the result, otherwise use cat + TODO: Move to a faster gzjoin-based technique + """ + #TODO: every time I try to import this from sniff, the parser dies + def is_gzip( filename ): + temp = open( filename, "U" ) + magic_check = temp.read( 2 ) + temp.close() + if magic_check != util.gzip_magic: + return False + return True + + if len(split_files) == 1: + os.system( 'mv -f %s %s' % ( split_files[0], output_file ) ) + return + if is_gzip(split_files[0]): + os.system( 'zcat %s | gzip -c > %s' % ( ' '.join(split_files), output_file ) ) + else: + os.system( 'cat %s > %s' % ( ' '.join(split_files), output_file ) ) + merge = staticmethod(merge) class Text( Data ): file_ext = 'txt' @@ -446,6 +472,80 @@ class Text( Data ): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disk' + def split( input_files, subdir_generator_function, split_params): + """ + Split the input files by line. + """ + if split_params is None: + return + + if len(input_files) > 1: + raise Exception("Text file splitting does not support multiple files") + + lines_per_file = None + chunk_size = None + if split_params['split_mode'] == 'number_of_parts': + lines_per_file = [] + # Computing the length is expensive! + def _file_len(fname): + i = 0 + f = open(fname) + for i, l in enumerate(f): + pass + f.close() + return i + 1 + length = _file_len(input_files[0]) + parts = int(split_params['split_size']) + if length < parts: + parts = length + len_each, remainder = divmod(length, parts) + while length > 0: + chunk = len_each + if remainder > 0: + chunk += 1 + lines_per_file.append(chunk) + remainder=- 1 + length -= chunk + elif split_params['split_mode'] == 'to_size': + chunk_size = int(split_params['split_size']) + else: + raise Exception('Unsupported split mode %s' % split_params['split_mode']) + + f = open(input_files[0], 'rt') + try: + chunk_idx = 0 + file_done = False + part_file = None + while not file_done: + if lines_per_file is None: + this_chunk_size = chunk_size + elif chunk_idx < len(lines_per_file): + this_chunk_size = lines_per_file[chunk_idx] + chunk_idx += 1 + lines_remaining = this_chunk_size + part_file = None + while lines_remaining > 0: + a_line = f.readline() + if a_line == '': + file_done = True + break + if part_file is None: + part_dir = subdir_generator_function() + part_path = os.path.join(part_dir, os.path.basename(input_files[0])) + part_file = open(part_path, 'w') + part_file.write(a_line) + lines_remaining -= 1 + if part_file is not None: + part_file.close() + except Exception, e: + log.error('Unable to split files: %s' % str(e)) + f.close() + if part_file is not None: + part_file.close() + raise + f.close() + split = staticmethod(split) + class Newick( Text ): pass diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 10780555af3..42bace94afc 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -49,6 +49,95 @@ class Sequence( data.Text ): else: dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disk' + + def split( input_files, subdir_generator_function, split_params): + """ + FASTQ files are split on cluster boundaries, in increments of 4 lines + """ + if split_params is None: + return + + def split_one( input_file, get_dir, clusters_per_file, default_clusters=None): + in_file = open(input_file, 'rt') + part_file = None + part = 0 + if clusters_per_file is None: + local_clusters_per_file = [default_clusters] + else: + local_clusters_per_file = [x for x in clusters_per_file] + + for i, line in enumerate(in_file): + cluster_number, line_in_cluster = divmod(i, 4) + current_part, remainder = divmod(cluster_number, local_clusters_per_file[part]) + + if (current_part != part or part_file is None): + if (part_file): + part_file.close() + part = current_part + part_dir = get_dir() + part_path = os.path.join(part_dir, os.path.basename(input_file)) + part_file = open(part_path, 'w') + if clusters_per_file is None: + local_clusters_per_file.append(default_clusters) + part_file.write(line) + if (part_file): + part_file.close() + in_file.close() + local_clusters_per_file[part] = remainder + 1 + return local_clusters_per_file + + directories = [] + def create_subdir(): + dir = subdir_generator_function() + directories.append(dir) + return dir + + clusters_per_file = None + if split_params['split_mode'] == 'number_of_parts': + # legacy splitting. To keep things simple, just scan the 0th file and count the clusters, + # then split it + clusters_per_file = [] + in_file = open(input_files[0], 'rt') + for i, line in enumerate(in_file): + pass + in_file.close() + length = (i+1)/4 + + if length <= 0: + raise Exception('Invalid sequence file %s' % input_files[0]) + parts = int(split_params['split_size']) + if length < parts: + parts = length + len_each, remainder = divmod(length, parts) + while length > 0: + chunk = len_each + if remainder > 0: + chunk += 1 + clusters_per_file.append(chunk) + remainder=- 1 + length -= chunk + split_one(input_files[0], create_subdir, clusters_per_file) + elif split_params['split_mode'] == 'to_size': + # split one file and see what the cluster sizes turn out to be + clusters_per_file = split_one(input_files[0], create_subdir, None, int(split_params['split_size'])) + else: + raise Exception('Unsupported split mode %s' % split_params['split_mode']) + + # split the rest, using the same number of clusters for each file + current_dir_idx = [0] # use a list to get around Python 2.x lame closure support + def get_subdir(): + if len(directories) <= current_dir_idx[0]: + raise Exception('FASTQ files do not have the same number of clusters - splitting failed') + result = directories[current_dir_idx[0]] + current_dir_idx[0] = current_dir_idx[0] + 1 + return result + + for i in range(1, len(input_files)): + current_dir_idx[0] = 0 + split_one(input_files[i], get_subdir, clusters_per_file) + split = staticmethod(split) + + class Alignment( data.Text ): """Class describing an alignment""" diff --git a/lib/galaxy/jobs/__init__.py b/lib/galaxy/jobs/__init__.py index 0ffc29ae223..64566d3a6d0 100644 --- a/lib/galaxy/jobs/__init__.py +++ b/lib/galaxy/jobs/__init__.py @@ -291,6 +291,7 @@ class JobWrapper( object ): self.working_directory = \ os.path.join( self.app.config.job_working_directory, str( self.job_id ) ) self.output_paths = None + self.output_dataset_paths = None self.tool_provided_job_metadata = None # Wrapper holding the info required to restore and clean up from files used for setting metadata externally self.external_output_metadata = metadata.JobExternalOutputMetadataWrapper( job ) @@ -657,23 +658,35 @@ class JobWrapper( object ): def get_session_id( self ): return self.session_id + def get_input_dataset_fnames( self, ds ): + filenames = [] + filenames = [ ds.file_name ] + #we will need to stage in metadata file names also + #TODO: would be better to only stage in metadata files that are actually needed (found in command line, referenced in config files, etc.) + for key, value in ds.metadata.items(): + if isinstance( value, model.MetadataFile ): + filenames.append( value.file_name ) + return filenames + def get_input_fnames( self ): job = self.get_job() filenames = [] for da in job.input_datasets: #da is JobToInputDatasetAssociation object if da.dataset: - filenames.append( da.dataset.file_name ) - #we will need to stage in metadata file names also - #TODO: would be better to only stage in metadata files that are actually needed (found in command line, referenced in config files, etc.) - for key, value in da.dataset.metadata.items(): - if isinstance( value, model.MetadataFile ): - filenames.append( value.file_name ) + filenames.extend(self.get_input_dataset_fnames(da.dataset)) return filenames def get_output_fnames( self ): - if self.output_paths is not None: - return self.output_paths + if self.output_paths is None: + self.compute_outputs() + return self.output_paths + def get_output_datasets_and_fnames( self ): + if self.output_dataset_paths is None: + self.compute_outputs() + return self.output_dataset_paths + + def compute_outputs( self ) : class DatasetPath( object ): def __init__( self, dataset_id, real_path, false_path = None ): self.dataset_id = dataset_id @@ -688,19 +701,25 @@ class JobWrapper( object ): job = self.get_job() # Job output datasets are combination of output datasets, library datasets, and jeha datasets. jeha = self.sa_session.query( model.JobExportHistoryArchive ).filter_by( job=job ).first() + jeha_false_path = None if self.app.config.outputs_to_working_directory: self.output_paths = [] + output_dataset_paths = {} for name, data in [ ( da.name, da.dataset.dataset ) for da in job.output_datasets + job.output_library_datasets ]: false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % data.id ) ) - self.output_paths.append( DatasetPath( data.id, data.file_name, false_path ) ) + dsp = DatasetPath( data.id, data.file_name, false_path ) + self.output_paths.append( dsp ) + self.output_dataset_paths[name] = data, dsp if jeha: - false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % jeha.dataset.id ) ) - self.output_paths.append( DatasetPath( jeha.dataset.id, jeha.dataset.file_name, false_path ) ) + jeha_false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % jeha.dataset.id ) ) else: - self.output_paths = [ DatasetPath( da.dataset.dataset.id, da.dataset.file_name ) for da in job.output_datasets + job.output_library_datasets ] - if jeha: - self.output_paths.append( DatasetPath( jeha.dataset.id, jeha.dataset.file_name ) ) - + results = [ (da.name, da.dataset, DatasetPath( da.dataset.dataset.id, da.dataset.file_name )) for da in job.output_datasets + job.output_library_datasets ] + self.output_paths = [t[2] for t in results] + self.output_dataset_paths = dict([(t[0], t[1:]) for t in results]) + + if jeha: + dsp = DatasetPath( jeha.dataset.id, jeha.dataset.file_name, jeha_false_path ) + self.output_paths.append( dsp ) return self.output_paths def get_output_file_id( self, file ): @@ -807,12 +826,7 @@ class TaskWrapper(JobWrapper): def __init__(self, task, queue): super(TaskWrapper, self).__init__(task.job, queue) self.task_id = task.id - self.parallelism = None - if task.part_file: - #do this better - self.working_directory = os.path.dirname(task.part_file) - else: - self.working_directory = None + self.working_directory = task.working_directory self.status = task.states.NEW def get_job( self ): diff --git a/lib/galaxy/jobs/runners/tasks.py b/lib/galaxy/jobs/runners/tasks.py index ff2238d2383..26726456002 100644 --- a/lib/galaxy/jobs/runners/tasks.py +++ b/lib/galaxy/jobs/runners/tasks.py @@ -60,71 +60,58 @@ class TaskedJobRunner( object ): if command_line: try: # DBTODO read tool info and use the right kind of parallelism. - # For now, the only splitter is the 'basic' one, n-ways split on one input, one output. - # This is incredibly simplified. Parallelism ultimately needs to describe which inputs, how, etc. + # For now, the only splitter is the 'basic' one job_wrapper.change_state( model.Job.states.RUNNING ) self.sa_session.flush() - parent_job = job_wrapper.get_job() # Split with the tool-defined method. - if job_wrapper.tool.parallelism == "basic": - from galaxy.jobs.splitters import basic - if len(job_wrapper.get_input_fnames()) > 1 or len(job_wrapper.get_output_fnames()) > 1: - log.error("The basic splitter is not capable of handling jobs with multiple inputs or outputs.") - job_wrapper.change_state( model.Job.states.ERROR ) - job_wrapper.fail("Job Splitting Failed, the basic splitter only handles tools with one input and one output") - # Requeue as a standard job? - return - input_file = job_wrapper.get_input_fnames()[0] - working_directory = job_wrapper.working_directory - # DBTODO execute an external task to do the splitting, this should happen at refactor. - # Regarding number of ways split, use "hints" in tool config? - # If the number of tasks is sufficiently high, we can use it to calculate job completion % and give a running status. - basic.split(input_file, working_directory, - 20, #Needs serious experimentation to find out what makes the most sense. - parent_job.input_datasets[0].dataset.ext) - # Tasks in this parts list are in alphabetical listdir order (15 before 5), but that should not matter. - parts = [os.path.join(os.path.abspath(job_wrapper.working_directory), p, os.path.basename(input_file)) - for p in os.listdir(job_wrapper.working_directory) - if p.startswith('task_')] - else: + try: + splitter = getattr(__import__('galaxy.jobs.splitters', globals(), locals(), [job_wrapper.tool.parallelism.method]), job_wrapper.tool.parallelism.method) + except: job_wrapper.change_state( model.Job.states.ERROR ) job_wrapper.fail("Job Splitting Failed, no match for '%s'" % job_wrapper.tool.parallelism) - # Assemble parts into task_wrappers + return + tasks = splitter.do_split(job_wrapper) # Not an option for now. Task objects don't *do* anything useful yet, but we'll want them tracked outside this thread to do anything. # if track_tasks_in_database: - tasks = [] task_wrappers = [] - for part in parts: - task = model.Task(parent_job, part) + for task in tasks: self.sa_session.add(task) - tasks.append(task) self.sa_session.flush() + # Must flush prior to the creation and queueing of task wrappers. for task in tasks: tw = TaskWrapper(task, job_wrapper.queue) task_wrappers.append(tw) self.app.job_manager.dispatcher.put(tw) tasks_incomplete = False + count_complete = 0 sleep_time = 1 + # sleep/loop until no more progress can be made. That is when + # all tasks are one of { OK, ERROR, DELETED } + completed_states = [ model.Task.states.OK, \ + model.Task.states.ERROR, \ + model.Task.states.DELETED ] + # TODO: Should we report an error (and not merge outputs) if one of the subtasks errored out? + # Should we prevent any that are pending from being started in that case? while tasks_incomplete is False: + count_complete = 0 tasks_incomplete = True for tw in task_wrappers: - if not tw.get_state() == model.Task.states.OK: + task_state = tw.get_state() + if not task_state in completed_states: tasks_incomplete = False - sleep( sleep_time ) - if sleep_time < 8: - sleep_time *= 2 - output_filename = job_wrapper.get_output_fnames()[0].real_path - basic.merge(working_directory, output_filename) - log.debug('execution finished: %s' % command_line) - for tw in task_wrappers: - # Prevent repetitive output, e.g. "Sequence File Aligned"x20 - # Eventually do a reduce for jobs that output "N reads mapped", combining all N for tasks. - if stdout.strip() != tw.get_task().stdout.strip(): - stdout += tw.get_task().stdout - if stderr.strip() != tw.get_task().stderr.strip(): - stderr += tw.get_task().stderr + else: + count_complete = count_complete + 1 + if tasks_incomplete is False: + # log.debug('Tasks complete: %s. Sleeping %s' % (count_complete, sleep_time)) + sleep( sleep_time ) + if sleep_time < 8: + sleep_time *= 2 + + log.debug('execution finished - beginning merge: %s' % command_line) + stdout, stderr = splitter.do_merge(job_wrapper, task_wrappers) + except Exception: job_wrapper.fail( "failure running job", exception=True ) log.exception("failure running job %d" % job_wrapper.job_id) diff --git a/lib/galaxy/jobs/splitters/basic.py b/lib/galaxy/jobs/splitters/basic.py index 4d1756c67e6..4ddbc83d429 100644 --- a/lib/galaxy/jobs/splitters/basic.py +++ b/lib/galaxy/jobs/splitters/basic.py @@ -1,91 +1,23 @@ -import os, logging +import logging +import multi + log = logging.getLogger( __name__ ) -def _file_len(fname): - i = 0 - f = open(fname) - for i, l in enumerate(f): - pass - f.close() - return i + 1 +def set_basic_defaults(job_wrapper): + parent_job = job_wrapper.get_job() + job_wrapper.tool.parallelism.attributes['split_inputs'] = parent_job.input_datasets[0].name + job_wrapper.tool.parallelism.attributes['merge_outputs'] = job_wrapper.get_output_datasets_and_fnames().keys()[0] -def _fq_seq_count(fname): - count = 0 - f = open(fname) - for i, l in enumerate(f): - if l.startswith('@'): - count += 1 - f.close() - return count - -def split_fq(input_file, working_directory, parts): - # Temporary, switch this to use the fq reader in lib/galaxy_utils/sequence. - outputs = [] - length = _fq_seq_count(input_file) - if length < 1: - return outputs - if length < parts: - parts = length - len_each, remainder = divmod(length, parts) - f = open(input_file, 'rt') - for p in range(0, parts): - part_dir = os.path.join( os.path.abspath(working_directory), 'task_%s' % p) - if not os.path.exists( part_dir ): - os.mkdir( part_dir ) - part_path = os.path.join(part_dir, os.path.basename(input_file)) - part_file = open(part_path, 'w') - for l in range(0, len_each): - part_file.write(f.readline()) - part_file.write(f.readline()) - part_file.write(f.readline()) - part_file.write(f.readline()) - if remainder > 0: - part_file.write(f.readline()) - part_file.write(f.readline()) - part_file.write(f.readline()) - part_file.write(f.readline()) - remainder -= 1 - outputs.append(part_path) - part_file.close() - f.close() - return outputs - -def split_txt(input_file, working_directory, parts): - outputs = [] - length = _file_len(input_file) - if length < parts: - parts = length - len_each, remainder = divmod(length, parts) - f = open(input_file, 'rt') - for p in range(0, parts): - part_dir = os.path.join( os.path.abspath(working_directory), 'task_%s' % p) - if not os.path.exists( part_dir ): - os.mkdir( part_dir ) - part_path = os.path.join(part_dir, os.path.basename(input_file)) - part_file = open(part_path, 'w') - for l in range(0, len_each): - part_file.write(f.readline()) - if remainder > 0: - part_file.write(f.readline()) - remainder -= 1 - outputs.append(part_path) - part_file.close() - f.close() - return outputs +def do_split (job_wrapper): + if len(job_wrapper.get_input_fnames()) > 1 or len(job_wrapper.get_output_fnames()) > 1: + log.error("The basic splitter is not capable of handling jobs with multiple inputs or outputs.") + raise Exception, "Job Splitting Failed, the basic splitter only handles tools with one input and one output" + # add in the missing information for splitting the one input and merging the one output + set_basic_defaults(job_wrapper) + return multi.do_split(job_wrapper) -def split( input_file, working_directory, parts, file_type = None): - #Implement a better method for determining how to split. - if file_type.startswith('fastq'): - return split_fq(input_file, working_directory, parts) - else: - return split_txt(input_file, working_directory, parts) +def do_merge( job_wrapper, task_wrappers): + # add in the missing information for splitting the one input and merging the one output + set_basic_defaults(job_wrapper) + return multi.do_merge(job_wrapper, task_wrappers) -def merge( working_directory, output_file ): - output_file_name = os.path.basename(output_file) - task_dirs = [os.path.join(working_directory, x) for x in os.listdir(working_directory) if x.startswith('task_')] - task_dirs.sort(key = lambda x: int(x.split('task_')[-1])) - for task_dir in task_dirs: - try: - os.system( 'cat %s >> %s' % ( os.path.join(task_dir, output_file_name), output_file ) ) - except Exception, e: - log.error(str(e)) diff --git a/lib/galaxy/jobs/splitters/multi.py b/lib/galaxy/jobs/splitters/multi.py new file mode 100644 index 00000000000..71df3058e76 --- /dev/null +++ b/lib/galaxy/jobs/splitters/multi.py @@ -0,0 +1,149 @@ +import os, logging, shutil +from galaxy import model + +log = logging.getLogger( __name__ ) + +def do_split (job_wrapper): + parent_job = job_wrapper.get_job() + working_directory = os.path.abspath(job_wrapper.working_directory) + + parallel_settings = job_wrapper.tool.parallelism.attributes + # Syntax: split_inputs="input1,input2" shared_inputs="genome" + # Designates inputs to be split or shared + split_inputs=parallel_settings.get("split_inputs") + if split_inputs is None: + split_inputs = [] + else: + split_inputs = [x.strip() for x in split_inputs.split(",")] + + shared_inputs=parallel_settings.get("shared_inputs") + if shared_inputs is None: + shared_inputs = [] + else: + shared_inputs = [x.strip() for x in shared_inputs.split(",")] + illegal_inputs = [x for x in shared_inputs if x in split_inputs] + if len(illegal_inputs) > 0: + raise Exception("Inputs have conflicting parallelism attributes: %s" % str( illegal_inputs )) + + subdir_index = [0] # use a list to get around Python 2.x lame closure support + task_dirs = [] + def get_new_working_directory_name(): + dir=os.path.join(working_directory, 'task_%d' % subdir_index[0]) + subdir_index[0] = subdir_index[0] + 1 + if not os.path.exists(dir): + os.makedirs(dir) + task_dirs.append(dir) + return dir + + # For things like paired end alignment, we need two inputs to be split. Since all inputs to all + # derived subtasks need to be correlated, allow only one input type to be split + type_to_input_map = {} + for input in parent_job.input_datasets: + if input.name in split_inputs: + type_to_input_map.setdefault(input.dataset.datatype, []).append(input.name) + elif input.name in shared_inputs: + pass # pass original file name + else: + log_error = "The input '%s' does not define a method for implementing parallelism" % str(input.name) + log.error(log_error) + raise Exception(log_error) + + if len(type_to_input_map) > 1: + log_error = "The multi splitter does not support splitting inputs of more than one type" + log.error(log_error) + raise Exception(log_error) + + # split the first one to build up the task directories + input_files = [] + for input in parent_job.input_datasets: + if input.name in split_inputs: + this_input_files = job_wrapper.get_input_dataset_fnames(input.dataset) + if len(this_input_files) > 1: + log_error = "The input '%s' is composed of multiple files - splitting is not allowed" % str(input.name) + log.error(log_error) + raise Exception(log_error) + input_files.extend(this_input_files) + + input_type = type_to_input_map.keys()[0] + # DBTODO execute an external task to do the splitting, this should happen at refactor. + # If the number of tasks is sufficiently high, we can use it to calculate job completion % and give a running status. + try: + input_type.split(input_files, get_new_working_directory_name, parallel_settings) + except AttributeError: + log_error = "The type '%s' does not define a method for splitting files" % str(input_type) + log.error(log_error) + raise + log.debug('do_split created %d parts' % len(task_dirs)) + # next, after we know how many divisions there are, add the shared inputs via soft links + for input in parent_job.input_datasets: + if input and input.name in shared_inputs: + names = job_wrapper.get_input_dataset_fnames(input.dataset) + for dir in task_dirs: + for file in names: + os.symlink(file, os.path.join(dir, os.path.basename(file))) + tasks = [] + for dir in task_dirs: + task = model.Task(parent_job, dir) + tasks.append(task) + return tasks + + +def do_merge( job_wrapper, task_wrappers): + parent_job = job_wrapper.get_job() + parallel_settings = job_wrapper.tool.parallelism.attributes + # Syntax: merge_outputs="export" pickone_outputs="genomesize" + # Designates outputs to be merged, or selected from as a representative + merge_outputs = parallel_settings.get("merge_outputs") + if merge_outputs is None: + merge_outputs = [] + else: + merge_outputs = [x.strip() for x in merge_outputs.split(",")] + pickone_outputs = parallel_settings.get("pickone_outputs") + if pickone_outputs is None: + pickone_outputs = [] + else: + pickone_outputs = [x.strip() for x in pickone_outputs.split(",")] + + illegal_outputs = [x for x in merge_outputs if x in pickone_outputs] + if len(illegal_outputs) > 0: + raise Exception("Outputs have conflicting parallelism attributes: %s" % str( illegal_outputs )) + + + working_directory = job_wrapper.working_directory + task_dirs = [os.path.join(working_directory, x) for x in os.listdir(working_directory) if x.startswith('task_')] + # TODO: Output datasets can be very complex. This doesn't handle metadata files + outputs = job_wrapper.get_output_datasets_and_fnames() + pickone_done = [] + task_dirs = [os.path.join(working_directory, x) for x in os.listdir(working_directory) if x.startswith('task_')] + for output in outputs: + output_file_name = str(outputs[output][1]) + base_output_name = os.path.basename(output_file_name) + if output in merge_outputs: + output_type = outputs[output][0].datatype + output_files = [os.path.join(dir,base_output_name) for dir in task_dirs] + log.debug('files %s ' % output_files) + output_type.merge(output_files, output_file_name) + log.debug('merge finished: %s' % output_file_name) + pass # TODO: merge all the files + elif output in pickone_outputs: + # just pick one of them + if output not in pickone_done: + task_file_name = os.path.join(task_dirs[0], base_output_name) + shutil.move( task_file_name, output_file_name ) + pickone_done.append(output) + else: + log_error = "The output '%s' does not define a method for implementing parallelism" % output + log.error(log_error) + raise Exception(log_error) + + stdout = '' + stderr='' + for tw in task_wrappers: + # Prevent repetitive output, e.g. "Sequence File Aligned"x20 + # Eventually do a reduce for jobs that output "N reads mapped", combining all N for tasks. + if stdout.strip() != tw.get_task().stdout.strip(): + stdout += tw.get_task().stdout + if stderr.strip() != tw.get_task().stderr.strip(): + stderr += tw.get_task().stderr + return (stdout, stderr) + diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 98448694c65..b2be39f43e2 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -176,12 +176,23 @@ class Task( object ): self.parameters = [] self.state = Task.states.NEW self.info = None + # TODO: Rename this to working_directory + # Does this necessitate a DB migration step? self.part_file = part_file self.task_runner_name = None self.task_runner_external_id = None self.job = job self.stdout = None self.stderr = None + + @property + def working_directory(self): + if self.part_file is not None: + if not os.path.isdir(self.part_file): + return os.path.dirname(self.part_file) + else: + return self.part_file + return None def set_state( self, state ): self.state = state diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index c03b2df4e33..da6a6fd3b08 100755 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -321,6 +321,21 @@ class ToolRequirement( object ): self.type = None self.version = None +class ToolParallelismInfo(object): + """ + Stores the information (if any) for running multiple instances of the tool in parallel + on the same set of inputs. + """ + def __init__(self, tag): + self.method = tag.get('method') + self.attributes = dict([item for item in tag.attrib.items() if item[0] != 'method' ]) + if len(self.attributes) == 0: + # legacy basic mode - provide compatible defaults + self.attributes['split_size'] = 20 + self.attributes['split_mode'] = 'number_of_parts' + + + class Tool: """ Represents a computational tool that can be executed through Galaxy. @@ -403,7 +418,7 @@ class Tool: # Parallelism for tasks, read from tool config. parallelism = root.find("parallelism") if parallelism is not None and parallelism.get("method"): - self.parallelism = parallelism.get("method") + self.parallelism = ToolParallelismInfo(parallelism) else: self.parallelism = None if self.app.config.start_job_runners is None: From 4c06b01d209aca07e947113434fa822047edbcd9 Mon Sep 17 00:00:00 2001 From: John Duddy Date: Wed, 25 May 2011 10:45:53 -0700 Subject: [PATCH 002/362] Remove zcat + gzip -c for merging gz files. Go with cat for now --- lib/galaxy/datatypes/data.py | 19 +++---------------- 1 file changed, 3 insertions(+), 16 deletions(-) diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index 9e0cf4c0b41..77983cc59dc 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -1,4 +1,4 @@ -import logging, os, sys, time, tempfile, gzip +import logging, os, sys, time, tempfile from galaxy import util from galaxy.util.odict import odict from galaxy.util.bunch import Bunch @@ -356,24 +356,11 @@ class Data( object ): def merge( split_files, output_file): """ - Export files are usually compressed, but it doesn't have to be so. In the case that they are, use - zcat to cat the files and gzip -c to recompress the result, otherwise use cat - TODO: Move to a faster gzjoin-based technique + TODO: Do we need to merge gzip files using gzjoin? cat seems to work, + but might be brittle. Need to revisit this. """ - #TODO: every time I try to import this from sniff, the parser dies - def is_gzip( filename ): - temp = open( filename, "U" ) - magic_check = temp.read( 2 ) - temp.close() - if magic_check != util.gzip_magic: - return False - return True - if len(split_files) == 1: os.system( 'mv -f %s %s' % ( split_files[0], output_file ) ) - return - if is_gzip(split_files[0]): - os.system( 'zcat %s | gzip -c > %s' % ( ' '.join(split_files), output_file ) ) else: os.system( 'cat %s > %s' % ( ' '.join(split_files), output_file ) ) merge = staticmethod(merge) From 97660b81d623b2dc41034d41bc7e604061380847 Mon Sep 17 00:00:00 2001 From: John Duddy Date: Thu, 16 Jun 2011 12:19:07 -0700 Subject: [PATCH 003/362] Add ability to split compressed input Fix off by one error on number_of_parts splitting --- lib/galaxy/datatypes/sequence.py | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 42bace94afc..0988cb64f35 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -2,6 +2,7 @@ Sequence classes """ +import gzip import data import logging import re @@ -58,17 +59,22 @@ class Sequence( data.Text ): return def split_one( input_file, get_dir, clusters_per_file, default_clusters=None): - in_file = open(input_file, 'rt') + compress = is_gzip(input_file) + if compress: +# TODO: Python 2.4, 2.5 don't have io.BufferedReader!!! +# add a buffered reader because gzip is really slow before python 2.7 + in_file = gzip.GzipFile(input_file, 'r') + else: + in_file = open(input_file, 'rt') part_file = None part = 0 if clusters_per_file is None: local_clusters_per_file = [default_clusters] else: local_clusters_per_file = [x for x in clusters_per_file] - for i, line in enumerate(in_file): cluster_number, line_in_cluster = divmod(i, 4) - current_part, remainder = divmod(cluster_number, local_clusters_per_file[part]) + current_part, remainder = divmod(cluster_number, local_clusters_per_file[part]+1) if (current_part != part or part_file is None): if (part_file): @@ -76,6 +82,7 @@ class Sequence( data.Text ): part = current_part part_dir = get_dir() part_path = os.path.join(part_dir, os.path.basename(input_file)) +# TODO: If the input was compressed, compress the output? part_file = open(part_path, 'w') if clusters_per_file is None: local_clusters_per_file.append(default_clusters) From 3d43bbc2e38e50356439ac9ab9173a8467644c8f Mon Sep 17 00:00:00 2001 From: John Duddy Date: Thu, 16 Jun 2011 14:06:06 -0700 Subject: [PATCH 004/362] Fix issues related to splitting multiple inputs and handling Files split evenly --- lib/galaxy/datatypes/sequence.py | 66 ++++++++++++++++++++++++++------ 1 file changed, 55 insertions(+), 11 deletions(-) diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 0988cb64f35..80e1324965e 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -58,7 +58,10 @@ class Sequence( data.Text ): if split_params is None: return - def split_one( input_file, get_dir, clusters_per_file, default_clusters=None): + def split_calculate_clusters( input_file, get_dir, default_clusters): + """ + Split the 0th file into even sized chunks, and return the number of clusters in each + """ compress = is_gzip(input_file) if compress: # TODO: Python 2.4, 2.5 don't have io.BufferedReader!!! @@ -68,13 +71,10 @@ class Sequence( data.Text ): in_file = open(input_file, 'rt') part_file = None part = 0 - if clusters_per_file is None: - local_clusters_per_file = [default_clusters] - else: - local_clusters_per_file = [x for x in clusters_per_file] + local_clusters_per_file = [] for i, line in enumerate(in_file): cluster_number, line_in_cluster = divmod(i, 4) - current_part, remainder = divmod(cluster_number, local_clusters_per_file[part]+1) + current_part, remainder = divmod(cluster_number, default_clusters) if (current_part != part or part_file is None): if (part_file): @@ -84,8 +84,7 @@ class Sequence( data.Text ): part_path = os.path.join(part_dir, os.path.basename(input_file)) # TODO: If the input was compressed, compress the output? part_file = open(part_path, 'w') - if clusters_per_file is None: - local_clusters_per_file.append(default_clusters) + local_clusters_per_file.append(default_clusters) part_file.write(line) if (part_file): part_file.close() @@ -93,6 +92,50 @@ class Sequence( data.Text ): local_clusters_per_file[part] = remainder + 1 return local_clusters_per_file + def split_to_size(input_file, get_dir, clusters_per_file): + """ + Split the files beyond the 0th to the same number of clusters as the 0th. + This is used to split in a variety of ways, so these are both legal for + clusters_per_file: + [ 10000, 10000, 10000, 10000, 2 ] # to_size=10000, 40002 total + [ 10001, 10001, 10000, 10000 ] # number_of_parts = 4, 40002 total + + """ + compress = is_gzip(input_file) + if compress: +# TODO: Python 2.4, 2.5 don't have io.BufferedReader!!! +# add a buffered reader because gzip is really slow before python 2.7 + in_file = gzip.GzipFile(input_file, 'r') + else: + in_file = open(input_file, 'rt') + part_file = None + part = 0 + clusters_this_part = 0 + for i, line in enumerate(in_file): + cluster_number, line_in_cluster = divmod(i, 4) + if clusters_this_part == clusters_per_file[part]: + current_part = part + 1 + else: + current_part = part + + if (current_part != part or part_file is None): + if (part_file): + part_file.close() + part = current_part + clusters_this_part = 0 + part_dir = get_dir() + part_path = os.path.join(part_dir, os.path.basename(input_file)) +# TODO: If the input was compressed, compress the output? + part_file = open(part_path, 'w') + if clusters_per_file is None and part > 0: + local_clusters_per_file.append(default_clusters) + part_file.write(line) + if line_in_cluster == 3: + clusters_this_part += 1 + if (part_file): + part_file.close() + in_file.close() + directories = [] def create_subdir(): dir = subdir_generator_function() @@ -123,10 +166,11 @@ class Sequence( data.Text ): clusters_per_file.append(chunk) remainder=- 1 length -= chunk - split_one(input_files[0], create_subdir, clusters_per_file) + split_to_size(input_files[0], create_subdir, clusters_per_file) elif split_params['split_mode'] == 'to_size': # split one file and see what the cluster sizes turn out to be - clusters_per_file = split_one(input_files[0], create_subdir, None, int(split_params['split_size'])) + clusters_per_file = split_calculate_clusters(input_files[0], create_subdir, + int(split_params['split_size'])) else: raise Exception('Unsupported split mode %s' % split_params['split_mode']) @@ -141,7 +185,7 @@ class Sequence( data.Text ): for i in range(1, len(input_files)): current_dir_idx[0] = 0 - split_one(input_files[i], get_subdir, clusters_per_file) + split_to_size(input_files[i], get_subdir, clusters_per_file) split = staticmethod(split) From aec48bd6a8f04cb3f7c1e4da00a3e96902b83c65 Mon Sep 17 00:00:00 2001 From: Enis Afgan Date: Tue, 5 Jul 2011 17:48:22 -0400 Subject: [PATCH 005/362] A very much in-progress code implementation of the ObjectStore - most of the functionality exists and works for interaction with a local file system and S3. Setting of the metadata does not work (empty files are created but never filled with content). Not sure if rerunning jobs with dependent datasets that have been deleted from cache works - some tools at least do. --- lib/galaxy/app.py | 6 +- lib/galaxy/config.py | 6 + lib/galaxy/datatypes/genetics.py | 2 +- lib/galaxy/jobs/__init__.py | 12 +- lib/galaxy/model/__init__.py | 100 +- lib/galaxy/model/mapping.py | 4 +- lib/galaxy/objectstore/__init__.py | 859 ++++++++++++++++++ lib/galaxy/objectstore/s3_multipart_upload.py | 85 ++ lib/galaxy/tools/__init__.py | 43 +- lib/galaxy/tools/actions/__init__.py | 3 +- lib/galaxy/tools/actions/upload.py | 1 - lib/galaxy/tools/actions/upload_common.py | 8 +- lib/galaxy/web/controllers/dataset.py | 42 +- lib/galaxy/web/controllers/history.py | 2 +- templates/dataset/display.mako | 18 +- templates/root/history.mako | 28 + templates/root/history_common.mako | 2 +- 17 files changed, 1127 insertions(+), 94 deletions(-) create mode 100644 lib/galaxy/objectstore/__init__.py create mode 100644 lib/galaxy/objectstore/s3_multipart_upload.py diff --git a/lib/galaxy/app.py b/lib/galaxy/app.py index 77e26a37d19..63c65059d99 100644 --- a/lib/galaxy/app.py +++ b/lib/galaxy/app.py @@ -7,6 +7,7 @@ from galaxy.web import security import galaxy.model import galaxy.datatypes.registry import galaxy.security +from galaxy.objectstore import build_object_store_from_config from galaxy.tags.tag_handler import GalaxyTagHandler from galaxy.tools.imp_exp import load_history_imp_exp_tools from galaxy.sample_tracking import external_service_types @@ -30,12 +31,15 @@ class UniverseApplication( object ): # Initialize database / check for appropriate schema version from galaxy.model.migrate.check import create_or_verify_database create_or_verify_database( db_url, kwargs.get( 'global_conf', {} ).get( '__file__', None ), self.config.database_engine_options ) + # Object store manager + self.object_store = build_object_store_from_config(self) # Setup the database engine and ORM from galaxy.model import mapping self.model = mapping.init( self.config.file_path, db_url, self.config.database_engine_options, - database_query_profiling_proxy = self.config.database_query_profiling_proxy ) + database_query_profiling_proxy = self.config.database_query_profiling_proxy, + object_store = self.object_store ) # Security helper self.security = security.SecurityHelper( id_secret=self.config.id_secret ) # Tag handler diff --git a/lib/galaxy/config.py b/lib/galaxy/config.py index ff69704f8a9..3987d495826 100644 --- a/lib/galaxy/config.py +++ b/lib/galaxy/config.py @@ -131,6 +131,12 @@ class Configuration( object ): self.nginx_upload_path = kwargs.get( 'nginx_upload_path', False ) if self.nginx_upload_store: self.nginx_upload_store = os.path.abspath( self.nginx_upload_store ) + self.object_store = kwargs.get( 'object_store', 'disk' ) + self.aws_access_key = kwargs.get( 'aws_access_key', None ) + self.aws_secret_key = kwargs.get( 'aws_secret_key', None ) + self.s3_bucket = kwargs.get( 's3_bucket', None) + self.use_reduced_redundancy = kwargs.get( 'use_reduced_redundancy', False ) + self.object_store_cache_size = float(kwargs.get( 'object_store_cache_size', -1 )) # Parse global_conf and save the parser global_conf = kwargs.get( 'global_conf', None ) global_conf_parser = ConfigParser.ConfigParser() diff --git a/lib/galaxy/datatypes/genetics.py b/lib/galaxy/datatypes/genetics.py index 5f9d8b4f712..8a5059b84cc 100644 --- a/lib/galaxy/datatypes/genetics.py +++ b/lib/galaxy/datatypes/genetics.py @@ -636,7 +636,7 @@ class RexpBase( Html ): def set_peek( self, dataset, **kwd ): """ expects a .pheno file in the extra_files_dir - ugh - note that R is wierd and does not include the row.name in + note that R is weird and does not include the row.name in the header. why?""" if not dataset.dataset.purged: pp = os.path.join(dataset.extra_files_path,'%s.pheno' % dataset.metadata.base_name) diff --git a/lib/galaxy/jobs/__init__.py b/lib/galaxy/jobs/__init__.py index 649e38eff7f..1f7ff18d6d2 100644 --- a/lib/galaxy/jobs/__init__.py +++ b/lib/galaxy/jobs/__init__.py @@ -208,7 +208,7 @@ class JobQueue( object ): log.error( "unknown job state '%s' for job %d" % ( job_state, job.id ) ) if not self.track_jobs_in_database: new_waiting_jobs.append( job.id ) - except Exception, e: + except Exception: log.exception( "failure running job %d" % job.id ) # Update the waiting list self.waiting_jobs = new_waiting_jobs @@ -332,7 +332,6 @@ class JobWrapper( object ): out_data = dict( [ ( da.name, da.dataset ) for da in job.output_datasets ] ) inp_data.update( [ ( da.name, da.dataset ) for da in job.input_library_datasets ] ) out_data.update( [ ( da.name, da.dataset ) for da in job.output_library_datasets ] ) - # Set up output dataset association for export history jobs. Because job # uses a Dataset rather than an HDA or LDA, it's necessary to set up a # fake dataset association that provides the needed attributes for @@ -428,6 +427,10 @@ class JobWrapper( object ): dataset.dataset.set_total_size() if dataset.ext == 'auto': dataset.extension = 'data' + # Update (non-library) job output datasets through the object store + if dataset not in job.output_library_datasets: + print "====== Handing failed job's dataset '%s' with name '%s' to object store" % (dataset.id, dataset.file_name) + self.app.object_store.update_from_file(dataset.id, create=True) self.sa_session.add( dataset ) self.sa_session.flush() job.state = model.Job.states.ERROR @@ -538,11 +541,14 @@ class JobWrapper( object ): else: # Security violation. log.exception( "from_work_dir specified a location not in the working directory: %s, %s" % ( source_file, self.working_directory ) ) - dataset.blurb = 'done' dataset.peek = 'no peek' dataset.info = context['stdout'] + context['stderr'] dataset.set_size() + # Update (non-library) job output datasets through the object store + if dataset not in job.output_library_datasets: + print "===+=== Handing dataset '%s' with name '%s' to object store" % (dataset.id, dataset.file_name) + self.app.object_store.update_from_file(dataset.id, create=True) if context['stderr']: dataset.blurb = "error" elif dataset.has_data(): diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 0219ff09256..493f64ce91e 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -522,6 +522,7 @@ class Dataset( object ): FAILED_METADATA = 'failed_metadata' ) permitted_actions = get_permitted_actions( filter='DATASET' ) file_path = "/tmp/" + object_store = None # This get initialized in mapping.py (method init) by app.py engine = None def __init__( self, id=None, state=None, external_filename=None, extra_files_path=None, file_size=None, purgable=True ): self.id = id @@ -535,17 +536,14 @@ class Dataset( object ): def get_file_name( self ): if not self.external_filename: assert self.id is not None, "ID must be set before filename used (commit the object)" - # First try filename directly under file_path - filename = os.path.join( self.file_path, "dataset_%d.dat" % self.id ) - # Only use that filename if it already exists (backward compatibility), - # otherwise construct hashed path - if not os.path.exists( filename ): - dir = os.path.join( self.file_path, *directory_hash_id( self.id ) ) + assert self.object_store is not None, "Object Store has not been initialized for dataset %s" % self.id + print "Calling get_filename 1", self.object_store + filename = self.object_store.get_filename( self.id ) + # print 'getting filename: ', filename + if not self.object_store.exists( self.id ): # Create directory if it does not exist - if not os.path.exists( dir ): - os.makedirs( dir ) - # Return filename inside hashed directory - return os.path.abspath( os.path.join( dir, "dataset_%d.dat" % self.id ) ) + self.object_store.create( self.id, dir_only=True ) + return filename else: filename = self.external_filename # Make filename absolute @@ -558,15 +556,8 @@ class Dataset( object ): file_name = property( get_file_name, set_file_name ) @property def extra_files_path( self ): - if self._extra_files_path: - path = self._extra_files_path - else: - path = os.path.join( self.file_path, "dataset_%d_files" % self.id ) - #only use path directly under self.file_path if it exists - if not os.path.exists( path ): - path = os.path.join( os.path.join( self.file_path, *directory_hash_id( self.id ) ), "dataset_%d_files" % self.id ) - # Make path absolute - return os.path.abspath( path ) + print "Calling get_filename 2", self.object_store + return self.object_store.get_filename( self.id, dir_only=True, extra_dir=self._extra_files_path or "dataset_%d_files" % self.id) def get_size( self, nice_size=False ): """Returns the size of the data on disk""" if self.file_size: @@ -575,20 +566,14 @@ class Dataset( object ): else: return self.file_size else: - try: - if nice_size: - return galaxy.datatypes.data.nice_size( os.path.getsize( self.file_name ) ) - else: - return os.path.getsize( self.file_name ) - except OSError: - return 0 + if nice_size: + return galaxy.datatypes.data.nice_size( self.object_store.size(self.id) ) + else: + return self.object_store.size(self.id) def set_size( self ): """Returns the size of the data on disk""" - try: - if not self.file_size: - self.file_size = os.path.getsize( self.file_name ) - except OSError: - self.file_size = 0 + if not self.file_size: + self.file_size = self.object_store.size(self.id) def get_total_size( self ): if self.total_size is not None: return self.total_size @@ -603,8 +588,9 @@ class Dataset( object ): if self.file_size is None: self.set_size() self.total_size = self.file_size or 0 - for root, dirs, files in os.walk( self.extra_files_path ): - self.total_size += sum( [ os.path.getsize( os.path.join( root, file ) ) for file in files ] ) + if self.object_store.exists(self.id, extra_dir=self._extra_files_path or "dataset_%d_files" % self.id, dir_only=True): + for root, dirs, files in os.walk( self.extra_files_path ): + self.total_size += sum( [ os.path.getsize( os.path.join( root, file ) ) for file in files ] ) def has_data( self ): """Detects whether there is any data""" return self.get_size() > 0 @@ -620,10 +606,7 @@ class Dataset( object ): # FIXME: sqlalchemy will replace this def _delete(self): """Remove the file that corresponds to this data""" - try: - os.remove(self.data.file_name) - except OSError, e: - log.critical('%s delete error %s' % (self.__class__.__name__, e)) + self.object_store.delete(self.id) @property def user_can_purge( self ): return self.purged == False \ @@ -631,9 +614,12 @@ class Dataset( object ): and len( self.history_associations ) == len( self.purged_history_associations ) def full_delete( self ): """Remove the file and extra files, marks deleted and purged""" - os.unlink( self.file_name ) - if os.path.exists( self.extra_files_path ): - shutil.rmtree( self.extra_files_path ) + # os.unlink( self.file_name ) + self.object_store.delete(self.id) + if self.object_store.exists(self.id, extra_dir=self._extra_files_path or "dataset_%d_files" % self.id, dir_only=True): + self.object_store.delete(self.id, entire_dir=True, extra_dir=self._extra_files_path or "dataset_%d_files" % self.id, dir_only=True) + # if os.path.exists( self.extra_files_path ): + # shutil.rmtree( self.extra_files_path ) # TODO: purge metadata files self.deleted = True self.purged = True @@ -1595,16 +1581,32 @@ class MetadataFile( object ): @property def file_name( self ): assert self.id is not None, "ID must be set before filename used (commit the object)" - path = os.path.join( Dataset.file_path, '_metadata_files', *directory_hash_id( self.id ) ) - # Create directory if it does not exist + # Ensure the directory structure and the metadata file object exist try: - os.makedirs( path ) - except OSError, e: - # File Exists is okay, otherwise reraise - if e.errno != errno.EEXIST: - raise - # Return filename inside hashed directory - return os.path.abspath( os.path.join( path, "metadata_%d.dat" % self.id ) ) + # self.history_dataset + # print "Dataset.file_path: %s, self.id: %s, self.history_dataset.dataset.object_store: %s" \ + # % (Dataset.file_path, self.id, self.history_dataset.dataset.object_store) + self.history_dataset.dataset.object_store.create( self.id, extra_dir='_metadata_files', extra_dir_at_root=True, alt_name="metadata_%d.dat" % self.id ) + print "Calling get_filename 3", self.object_store + path = self.history_dataset.dataset.object_store.get_filename( self.id, extra_dir='_metadata_files', extra_dir_at_root=True, alt_name="metadata_%d.dat" % self.id ) + print "Created metadata file at path: %s" % path + self.library_dataset + # raise + return path + except AttributeError: + # In case we're not working with the history_dataset + # print "Caught AttributeError" + path = os.path.join( Dataset.file_path, '_metadata_files', *directory_hash_id( self.id ) ) + # Create directory if it does not exist + try: + os.makedirs( path ) + except OSError, e: + # File Exists is okay, otherwise reraise + if e.errno != errno.EEXIST: + raise + # Return filename inside hashed directory + return os.path.abspath( os.path.join( path, "metadata_%d.dat" % self.id ) ) + class FormDefinition( object, APIItem ): # The following form_builder classes are supported by the FormDefinition class. diff --git a/lib/galaxy/model/mapping.py b/lib/galaxy/model/mapping.py index 06c1ae2d245..7f0837682ce 100644 --- a/lib/galaxy/model/mapping.py +++ b/lib/galaxy/model/mapping.py @@ -1709,10 +1709,12 @@ def load_egg_for_url( url ): # Let this go, it could possibly work with db's we don't support log.error( "database_connection contains an unknown SQLAlchemy database dialect: %s" % dialect ) -def init( file_path, url, engine_options={}, create_tables=False, database_query_profiling_proxy=False ): +def init( file_path, url, engine_options={}, create_tables=False, database_query_profiling_proxy=False, object_store=None ): """Connect mappings to the database""" # Connect dataset to the file path Dataset.file_path = file_path + # Connect dataset to object store + Dataset.object_store = object_store # Load the appropriate db module load_egg_for_url( url ) # Should we use the logging proxy? diff --git a/lib/galaxy/objectstore/__init__.py b/lib/galaxy/objectstore/__init__.py new file mode 100644 index 00000000000..076ca8cfbd1 --- /dev/null +++ b/lib/galaxy/objectstore/__init__.py @@ -0,0 +1,859 @@ +""" +objectstore package, abstraction for storing blobs of data for use in Galaxy, +all providers ensure that data can be accessed on the filesystem for running +tools +""" + +import os +import time +import shutil +import logging +import threading +import subprocess +import multiprocessing +from datetime import datetime + +from galaxy import util +from galaxy.jobs import Sleeper +from galaxy.model import directory_hash_id +from galaxy.objectstore.s3_multipart_upload import multipart_upload + +from boto.s3.key import Key +from boto.s3.connection import S3Connection +from boto.exception import S3ResponseError + +log = logging.getLogger( __name__ ) +logging.getLogger('boto').setLevel(logging.INFO) # Otherwise boto is quite noisy + + +class ObjectNotFound(Exception): + """ Accessed object was not found """ + pass + + +class ObjectStore(object): + """ + ObjectStore abstract interface + """ + def __init__(self): + self.running = True + + def shutdown(self): + self.running = False + + def exists(self, dataset_id, dir_only=False, extra_dir=None, extra_dir_at_root=False, alt_name=None): + """ + Returns True if the object identified by `dataset_id` exists in this + file store, False otherwise. + + FIELD DESCRIPTIONS (these apply to all the methods in this class): + :type dataset_id: int + :param dataset_id: Galaxy-assigned database ID of the dataset to be checked. + + :type dir_only: bool + :param dir_only: If True, check only the path where the file + identified by `dataset_id` should be located, not the + dataset itself. This option applies to `extra_dir` + argument as well. + + :type extra_dir: string + :param extra_dir: Append `extra_dir` to the directory structure where + the dataset identified by `dataset_id` should be located. + (e.g., 000/extra_dir/dataset_id) + + :type extra_dir_at_root: bool + :param extra_dir_at_root: Applicable only if `extra_dir` is set. + If True, the `extra_dir` argument is placed at + root of the created directory structure rather + than at the end (e.g., extra_dir/000/dataset_id + vs. 000/extra_dir/dataset_id) + + :type alt_name: string + :param alt_name: Use this name as the alternative name for the created + dataset rather than the default. + """ + raise NotImplementedError() + + def file_ready(self, dataset_id, dir_only=False, extra_dir=None, extra_dir_at_root=False, alt_name=None): + """ A helper method that checks if a file corresponding to a dataset + is ready and available to be used. Return True if so, False otherwise.""" + return True + + def create(self, dataset_id, dir_only=False, extra_dir=None, extra_dir_at_root=False, alt_name=None): + """ + Mark the object identified by `dataset_id` as existing in the store, but + with no content. This method will create a proper directory structure for + the file if the directory does not already exist. + See `exists` method for the description of the fields. + """ + raise NotImplementedError() + + def empty(self, dataset_id, extra_dir=None, extra_dir_at_root=False, alt_name=None): + """ + Test if the object identified by `dataset_id` has content. + If the object does not exist raises `ObjectNotFound`. + See `exists` method for the description of the fields. + """ + raise NotImplementedError() + + def size(self, dataset_id, extra_dir=None, extra_dir_at_root=False, alt_name=None): + """ + Return size of the object identified by `dataset_id`. + If the object does not exist, return 0. + See `exists` method for the description of the fields. + """ + raise NotImplementedError() + + def delete(self, dataset_id, entire_dir=False, extra_dir=None, extra_dir_at_root=False, alt_name=None): + """ + Deletes the object identified by `dataset_id`. + See `exists` method for the description of other fields. + :type entire_dir: bool + :param entire_dir: If True, delete the entire directory pointed to by + extra_dir. For safety reasons, this option applies + only for and in conjunction with the extra_dir option. + """ + raise NotImplementedError() + + def get_data(self, dataset_id, start=0, count=-1, extra_dir=None, extra_dir_at_root=False, alt_name=None): + """ + Fetch `count` bytes of data starting at offset `start` from the + object identified uniquely by `dataset_id`. + If the object does not exist raises `ObjectNotFound`. + See `exists` method for the description of other fields. + + :type start: int + :param start: Set the position to start reading the dataset file + + :type count: int + :param count: Read at most `count` bytes from the dataset + """ + raise NotImplementedError() + + def get_filename(self, dataset_id, dir_only=False, extra_dir=None, extra_dir_at_root=False, alt_name=None): + """ + Get the expected filename (including the absolute path) which can be used + to access the contents of the object uniquely identified by `dataset_id`. + See `exists` method for the description of the fields. + """ + raise NotImplementedError() + + def update_from_file(self, dataset_id, extra_dir=None, extra_dir_at_root=False, alt_name=None, filename=None, create=False): + """ + Inform the store that the file associated with the object has been + updated. If `filename` is provided, update from that file instead + of the default. + If the object does not exist raises `ObjectNotFound`. + See `exists` method for the description of other fields. + + :type filename: string + :param filename: Use file pointed to by `filename` as the source for + updating the dataset identified by `dataset_id` + + :type create: bool + :param create: If True and the default dataset does not exist, create it first. + """ + raise NotImplementedError() + + def get_object_url(self, dataset_id, extra_dir=None, extra_dir_at_root=False, alt_name=None): + """ + If the store supports direct URL access, return a URL. Otherwise return + None. + Note: need to be careful to to bypass dataset security with this. + See `exists` method for the description of the fields. + """ + raise NotImplementedError() + + ## def get_staging_command( id ): + ## """ + ## Return a shell command that can be prepended to the job script to stage the + ## dataset -- runs on worker nodes. + ## + ## Note: not sure about the interface here. Should this return a filename, command + ## tuple? Is this even a good idea, seems very useful for S3, other object stores? + ## """ + + +class DiskObjectStore(ObjectStore): + """ + Standard Galaxy object store, stores objects in files under a specific + directory on disk. + """ + def __init__(self, app): + super(DiskObjectStore, self).__init__() + self.file_path = app.config.file_path + + def _get_filename(self, dataset_id, dir_only=False, extra_dir=None, extra_dir_at_root=False, alt_name=None): + """Class method that returns the absolute path for the file corresponding + to the `dataset_id` regardless of whether the file exists. + """ + path = self._construct_path(dataset_id, dir_only=dir_only, extra_dir=extra_dir, extra_dir_at_root=extra_dir_at_root, alt_name=alt_name, old_style=True) + # For backward compatibility, check the old style root path first; otherwise, + # construct hashed path + if not os.path.exists(path): + return self._construct_path(dataset_id, dir_only=dir_only, extra_dir=extra_dir, extra_dir_at_root=extra_dir_at_root, alt_name=alt_name) + + def _construct_path(self, dataset_id, old_style=False, dir_only=False, extra_dir=None, extra_dir_at_root=False, alt_name=None): + """ Construct the expected absolute path for accessing the object + identified by `dataset_id`. + + :type dir_only: bool + :param dir_only: If True, return only the absolute path where the file + identified by `dataset_id` should be located + + :type extra_dir: string + :param extra_dir: Append the value of this parameter to the expected path + used to access the object identified by `dataset_id` + (e.g., /files/000//dataset_10.dat). + + :type alt_name: string + :param alt_name: Use this name as the alternative name for the returned + dataset rather than the default. + + :type old_style: bool + param old_style: This option is used for backward compatibility. If True + the composed directory structure does not include a hash id + (e.g., /files/dataset_10.dat (old) vs. /files/000/dataset_10.dat (new)) + """ + if old_style: + if extra_dir is not None: + path = os.path.join(self.file_path, extra_dir) + else: + path = self.file_path + else: + rel_path = os.path.join(*directory_hash_id(dataset_id)) + if extra_dir is not None: + if extra_dir_at_root: + rel_path = os.path.join(extra_dir, rel_path) + else: + rel_path = os.path.join(rel_path, extra_dir) + path = os.path.join(self.file_path, rel_path) + if not dir_only: + path = os.path.join(path, alt_name if alt_name else "dataset_%s.dat" % dataset_id) + return os.path.abspath(path) + + def exists(self, dataset_id, **kwargs): + path = self._construct_path(dataset_id, old_style=True, **kwargs) + # For backward compatibility, check root path first; otherwise, construct + # and check hashed path + if not os.path.exists(path): + path = self._construct_path(dataset_id, **kwargs) + return os.path.exists(path) + + def create(self, dataset_id, **kwargs): + if not self.exists(dataset_id, **kwargs): + # Pull out locally used fields + extra_dir = kwargs.get('extra_dir', None) + extra_dir_at_root = kwargs.get('extra_dir_at_root', False) + dir_only = kwargs.get('dir_only', False) + alt_name = kwargs.get('alt_name', None) + # Construct hashed path + path = os.path.join(*directory_hash_id(dataset_id)) + # Optionally append extra_dir + if extra_dir is not None: + if extra_dir_at_root: + path = os.path.join(extra_dir, path) + else: + path = os.path.join(path, extra_dir) + # Combine the constructted path with the root dir for all files + path = os.path.join(self.file_path, path) + # Create directory if it does not exist + if not os.path.exists(path): + os.makedirs(path) + if not dir_only: + path = os.path.join(path, alt_name if alt_name else "dataset_%s.dat" % dataset_id) + open(path, 'w').close() + + def empty(self, dataset_id, **kwargs): + return os.path.getsize(self.get_filename(dataset_id, **kwargs)) > 0 + + def size(self, dataset_id, **kwargs): + if self.exists(dataset_id, **kwargs): + try: + return os.path.getsize(self.get_filename(dataset_id, **kwargs)) + except OSError: + return 0 + else: + return 0 + + def delete(self, dataset_id, entire_dir=False, **kwargs): + path = self.get_filename(dataset_id, **kwargs) + extra_dir = kwargs.get('extra_dir', None) + try: + if entire_dir and extra_dir: + shutil.rmtree(path) + return True + if self.exists(dataset_id, **kwargs): + os.remove(path) + return True + except OSError, ex: + log.critical('%s delete error %s' % (self._get_filename(dataset_id, **kwargs), ex)) + return False + + def get_data(self, dataset_id, start=0, count=-1, **kwargs): + data_file = open(self.get_filename(dataset_id, **kwargs), 'r') + data_file.seek(start) + content = data_file.read(count) + data_file.close() + return content + + def get_filename(self, dataset_id, **kwargs): + path = self._construct_path(dataset_id, old_style=True, **kwargs) + # For backward compatibility, check root path first; otherwise, construct + # and check hashed path + if os.path.exists(path): + return path + else: + path = self._construct_path(dataset_id, **kwargs) + print "Checking it %s exists: %s" %(path, os.path.exists(path)) + if os.path.exists(path): + return path + else: + raise ObjectNotFound() + + def update_from_file(self, dataset_id, file_name=None, create=False, **kwargs): + """ `create` parameter is not used in this implementation """ + if create: + self.create(dataset_id, **kwargs) + if file_name and self.exists(dataset_id, **kwargs): + try: + shutil.copy(file_name, self.get_filename(dataset_id, **kwargs)) + except IOError, ex: + log.critical('Error copying %s to %s: %s' % (file_name, + self._get_filename(dataset_id, **kwargs), ex)) + + def get_object_url(self, dataset_id, **kwargs): + return None + + + +class CachingObjectStore(ObjectStore): + """ + Object store that uses a directory for caching files, but defers and writes + back to another object store. + """ + + def __init__(self, path, backend): + super(CachingObjectStore, self).__init__(self, path, backend) + + + +class S3ObjectStore(ObjectStore): + """ + Object store that stores objects as items in an AWS S3 bucket. A local + cache exists that is used as an intermediate location for files between + Galaxy and S3. + """ + def __init__(self, app): + super(S3ObjectStore, self).__init__() + self.app = app + self.staging_path = self.app.config.file_path + self.s3_conn = S3Connection() + self.bucket = self._get_bucket(self.app.config.s3_bucket) + self.use_rr = self.app.config.use_reduced_redundancy + self.cache_size = self.app.config.object_store_cache_size * 1073741824 # Convert GBs to bytes + self.transfer_progress = 0 + # Clean cache only if value is set in universe_wsgi.ini + if self.cache_size != -1: + # Helper for interruptable sleep + self.sleeper = Sleeper() + self.cache_monitor_thread = threading.Thread(target=self.__cache_monitor) + self.cache_monitor_thread.start() + log.info("Cache cleaner manager started") + + def __cache_monitor(self): + time.sleep(2) # Wait for things to load before starting the monitor + while self.running: + total_size = 0 + # Is this going to be too expensive of an operation to be done frequently? + file_list = [] + for dirpath, dirnames, filenames in os.walk(self.staging_path): + for f in filenames: + fp = os.path.join(dirpath, f) + file_size = os.path.getsize(fp) + total_size += file_size + # Get the time given file was last accessed + last_access_time = time.localtime(os.stat(fp)[7]) + # Compose a tuple of the access time and the file path + file_tuple = last_access_time, fp, file_size + file_list.append(file_tuple) + # Sort the file list (based on access time) + file_list.sort() + # Initiate cleaning once within 10% of the defined cache size? + cache_limit = self.cache_size * 0.9 + if total_size > cache_limit: + log.info("Initiating cache cleaning: current cache size: %s; clean until smaller than: %s" \ + % (convert_bytes(total_size), convert_bytes(cache_limit))) + # How much to delete? If simply deleting up to the cache-10% limit, + # is likely to be deleting frequently and may run the risk of hitting + # the limit - maybe delete additional #%? + # For now, delete enough to leave at least 10% of the total cache free + delete_this_much = total_size - cache_limit + self.__clean_cache(file_list, delete_this_much) + self.sleeper.sleep(30) # Test cache size every 30 seconds? + + def __clean_cache(self, file_list, delete_this_much): + """ Keep deleting files from the file_list until the size of the deleted + files is greater than the value in delete_this_much parameter. + + :type file_list: list + :param file_list: List of candidate files that can be deleted. This method + will start deleting files from the beginning of the list so the list + should be sorted accordingly. The list must contains 3-element tuples, + positioned as follows: position 0 holds file last accessed timestamp + (as time.struct_time), position 1 holds file path, and position 2 has + file size (e.g., (, /mnt/data/dataset_1.dat), 472394) + + :type delete_this_much: int + :param delete_this_much: Total size of files, in bytes, that should be deleted. + """ + # Keep deleting datasets from file_list until deleted_amount does not + # exceed delete_this_much; start deleting from the front of the file list, + # which assumes the oldest files come first on the list. + deleted_amount = 0 + for i, f in enumerate(file_list): + if deleted_amount < delete_this_much: + deleted_amount += f[2] + os.remove(f[1]) + # Debugging code for printing deleted files' stats + # folder, file_name = os.path.split(f[1]) + # file_date = time.strftime("%m/%d/%y %H:%M:%S", f[0]) + # log.debug("%s. %-25s %s, size %s (deleted %s/%s)" \ + # % (i, file_name, convert_bytes(f[2]), file_date, \ + # convert_bytes(deleted_amount), convert_bytes(delete_this_much))) + else: + log.debug("Cache cleaning done. Total space freed: %s" % convert_bytes(deleted_amount)) + return + + def _get_bucket(self, bucket_name): + """ Sometimes a handle to a bucket is not established right away so try + it a few times. Raise error is connection is not established. """ + for i in range(5): + try: + bucket = self.s3_conn.get_bucket(bucket_name) + log.debug("Using S3 object store; got bucket '%s'" % bucket.name) + return bucket + except S3ResponseError: + log.debug("Could not get bucket '%s', attempt %s/5" % (bucket_name, i+1)) + time.sleep(2) + # All the attempts have been exhausted and connection was not established, + # raise error + raise S3ResponseError + + def _fix_permissions(self, rel_path): + """ Set permissions on rel_path""" + for basedir, dirs, files in os.walk( rel_path ): + util.umask_fix_perms( basedir, self.app.config.umask, 0777, self.app.config.gid ) + for f in files: + path = os.path.join( basedir, f ) + # Ignore symlinks + if os.path.islink( path ): + continue + util.umask_fix_perms( path, self.app.config.umask, 0666, self.app.config.gid ) + + def _construct_path(self, dataset_id, dir_only=None, extra_dir=None, extra_dir_at_root=False, alt_name=None): + rel_path = os.path.join(*directory_hash_id(dataset_id)) + if extra_dir is not None: + if extra_dir_at_root: + rel_path = os.path.join(extra_dir, rel_path) + else: + rel_path = os.path.join(rel_path, extra_dir) + # S3 folders are marked by having trailing '/' so add it now + rel_path = '%s/' % rel_path + if not dir_only: + rel_path = os.path.join(rel_path, alt_name if alt_name else "dataset_%s.dat" % dataset_id) + return rel_path + + def _get_cache_path(self, rel_path): + return os.path.abspath(os.path.join(self.staging_path, rel_path)) + + def _get_transfer_progress(self): + return self.transfer_progress + + def _get_size_in_s3(self, rel_path): + try: + key = self.bucket.get_key(rel_path) + if key: + return key.size + except S3ResponseError, ex: + log.error("Could not get size of key '%s' from S3: %s" % (rel_path, ex)) + except Exception, ex: + log.error("Could not get reference to the key object '%s'; returning -1 for key size: %s" % (rel_path, ex)) + return -1 + + def _key_exists(self, rel_path): + exists = False + try: + # A hackish way of testing if the rel_path is a folder vs a file + is_dir = rel_path[-1] == '/' + if is_dir: + rs = self.bucket.get_all_keys(prefix=rel_path) + if len(rs) > 0: + exists = True + else: + exists = False + else: + key = Key(self.bucket, rel_path) + exists = key.exists() + except S3ResponseError, ex: + log.error("Trouble checking existence of S3 key '%s': %s" % (rel_path, ex)) + return False + print "Checking if '%s' exists in S3: %s" % (rel_path, exists) + if rel_path[0] == '/': + raise + return exists + + def _in_cache(self, rel_path): + """ Check if the given dataset is in the local cache and return True if so. """ + # log.debug("------ Checking cache for rel_path %s" % rel_path) + cache_path = self._get_cache_path(rel_path) + exists = os.path.exists(cache_path) + # print "Checking chache for %s; returning %s" % (cache_path, exists) + return exists + # EATODO: Part of checking if a file is in cache should be to ensure the + # size of the cached file matches that on S3. Once the upload tool explicitly + # creates, this check sould be implemented- in the mean time, it's not + # looking likely to be implementable reliably. + # if os.path.exists(cache_path): + # # print "***1 %s exists" % cache_path + # if self._key_exists(rel_path): + # # print "***2 %s exists in S3" % rel_path + # # Make sure the size in cache is available in its entirety + # # print "File '%s' cache size: %s, S3 size: %s" % (cache_path, os.path.getsize(cache_path), self._get_size_in_s3(rel_path)) + # if os.path.getsize(cache_path) == self._get_size_in_s3(rel_path): + # # print "***2.1 %s exists in S3 and the size is the same as in cache (in_cache=True)" % rel_path + # exists = True + # else: + # # print "***2.2 %s exists but differs in size from cache (in_cache=False)" % cache_path + # exists = False + # else: + # # Although not perfect decision making, this most likely means + # # that the file is currently being uploaded + # # print "***3 %s found in cache but not in S3 (in_cache=True)" % cache_path + # exists = True + # else: + # # print "***4 %s does not exist (in_cache=False)" % cache_path + # exists = False + # # print "Checking cache for %s; returning %s" % (cache_path, exists) + # return exists + # # return False + + def _pull_into_cache(self, rel_path): + # Ensure the cache directory structure exists (e.g., dataset_#_files/) + rel_path_dir = os.path.dirname(rel_path) + if not os.path.exists(self._get_cache_path(rel_path_dir)): + os.makedirs(self._get_cache_path(rel_path_dir)) + # Now pull in the file + ok = self._download(rel_path) + self._fix_permissions(rel_path) + return ok + + def _transfer_cb(self, complete, total): + self.transfer_progress += 10 + # print "Dataset transfer progress: %s" % self.transfer_progress + + def _download(self, rel_path): + try: + log.debug("Pulling key '%s' into cache to %s" % (rel_path, self._get_cache_path(rel_path))) + key = self.bucket.get_key(rel_path) + # Test is cache is large enough to hold the new file + if key.size > self.cache_size: + log.critical("File %s is larger (%s) than the cache size (%s). Cannot download." \ + % (rel_path, key.size, self.cache_size)) + return False + # Test if 'axel' is available for parallel download and pull the key into cache + try: + ret_code = subprocess.call('axel') + except OSError: + ret_code = 127 + if ret_code == 127: + self.transfer_progress = 0 # Reset transfer progress counter + key.get_contents_to_filename(self._get_cache_path(rel_path), cb=self._transfer_cb, num_cb=10) + print "(ssss) Pulled key '%s' into cache to %s" % (rel_path, self._get_cache_path(rel_path)) + return True + else: + ncores = multiprocessing.cpu_count() + url = key.generate_url(7200) + ret_code = subprocess.call("axel -a -n %s '%s'" % (ncores, url)) + if ret_code == 0: + print "(ssss) Parallel pulled key '%s' into cache to %s" % (rel_path, self._get_cache_path(rel_path)) + return True + except S3ResponseError, ex: + log.error("Problem downloading key '%s' from S3 bucket '%s': %s" % (rel_path, self.bucket.name, ex)) + return False + + def _push_to_s3(self, rel_path, source_file=None, from_string=None): + """ + Push the file pointed to by `rel_path` to S3 naming the key `rel_path`. + If `source_file` is provided, push that file instead while still using + `rel_path` as the key name. + If `from_string` is provided, set contents of the file to the value of + the string + """ + try: + source_file = source_file if source_file else self._get_cache_path(rel_path) + if os.path.exists(source_file): + key = Key(self.bucket, rel_path) + if from_string: + key.set_contents_from_string(from_string, reduced_redundancy=self.use_rr) + log.debug("Pushed data from string '%s' to key '%s'" % (from_string, rel_path)) + else: + start_time = datetime.now() + print "[%s] Pushing cache file '%s' to key '%s'" % (start_time, source_file, rel_path) + mb_size = os.path.getsize(source_file) / 1e6 + if mb_size < 60: + self.transfer_progress = 0 # Reset transfer progress counter + key.set_contents_from_filename(source_file, reduced_redundancy=self.use_rr, + cb=self._transfer_cb, num_cb=10) + else: + multipart_upload(self.bucket, key.name, source_file, mb_size, use_rr=self.use_rr) + # self._multipart_upload(key.name, source_file, mb_size) + end_time = datetime.now() + print "Push ended at '%s'; it lasted '%s'" % (end_time, end_time-start_time) + log.debug("Pushed cache file '%s' to key '%s'" % (source_file, rel_path)) + return True + else: + log.error("Tried updating key '%s' from source file '%s', but source file does not exist." + % (rel_path, source_file)) + except S3ResponseError, ex: + log.error("Trouble pushing S3 key '%s' from file '%s': %s" % (rel_path, source_file, ex)) + return False + + def file_ready(self, dataset_id, **kwargs): + """ A helper method that checks if a file corresponding to a dataset + is ready and available to be used. Return True if so, False otherwise.""" + rel_path = self._construct_path(dataset_id, **kwargs) + # Make sure the size in cache is available in its entirety + if self._in_cache(rel_path) and os.path.getsize(self._get_cache_path(rel_path)) == self._get_size_in_s3(rel_path): + return True + return False + + def exists(self, dataset_id, **kwargs): + in_cache = in_s3 = False + rel_path = self._construct_path(dataset_id, **kwargs) + # Check cache + if self._in_cache(rel_path): + in_cache = True + # Check S3 + in_s3 = self._key_exists(rel_path) + # log.debug("~~~~~~ File '%s' exists in cache: %s; in s3: %s" % (rel_path, in_cache, in_s3)) + # dir_only does not get synced so shortcut the decision + dir_only = kwargs.get('dir_only', False) + if dir_only: + if in_cache or in_s3: + return True + else: + return False + # TODO: Sync should probably not be done here. Add this to an async upload stack? + if in_cache and not in_s3: + self._push_to_s3(rel_path, source_file=self._get_cache_path(rel_path)) + return True + elif in_s3: + return True + else: + return False + + def create(self, dataset_id, **kwargs): + if not self.exists(dataset_id, **kwargs): + print "S3 OS creating a dataset with ID %s" % dataset_id + # Pull out locally used fields + extra_dir = kwargs.get('extra_dir', None) + extra_dir_at_root = kwargs.get('extra_dir_at_root', False) + dir_only = kwargs.get('dir_only', False) + alt_name = kwargs.get('alt_name', None) + # print "---- Processing: %s; %s" % (alt_name, locals()) + # Construct hashed path + rel_path = os.path.join(*directory_hash_id(dataset_id)) + # Optionally append extra_dir + if extra_dir is not None: + if extra_dir_at_root: + rel_path = os.path.join(extra_dir, rel_path) + else: + rel_path = os.path.join(rel_path, extra_dir) + # Create given directory in cache + cache_dir = os.path.join(self.staging_path, rel_path) + if not os.path.exists(cache_dir): + os.makedirs(cache_dir) + # Although not really necessary to create S3 folders (because S3 has + # flat namespace), do so for consistency with the regular file system + # S3 folders are marked by having trailing '/' so add it now + # s3_dir = '%s/' % rel_path + # self._push_to_s3(s3_dir, from_string='') + # If instructed, create the dataset in cache & in S3 + if not dir_only: + rel_path = os.path.join(rel_path, alt_name if alt_name else "dataset_%s.dat" % dataset_id) + open(os.path.join(self.staging_path, rel_path), 'w').close() + self._push_to_s3(rel_path, from_string='') + + def empty(self, dataset_id, **kwargs): + if self.exists(dataset_id, **kwargs): + return bool(self.size(dataset_id, **kwargs) > 0) + else: + raise ObjectNotFound() + + def size(self, dataset_id, **kwargs): + rel_path = self._construct_path(dataset_id, **kwargs) + if self._in_cache(rel_path): + try: + return os.path.getsize(self._get_cache_path(rel_path)) + except OSError, ex: + log.info("Could not get size of file '%s' in local cache, will try S3. Error: %s" % (rel_path, ex)) + elif self.exists(dataset_id, **kwargs): + return self._get_size_in_s3(rel_path) + log.warning("Did not find dataset '%s', returning 0 for size" % rel_path) + return 0 + + def delete(self, dataset_id, entire_dir=False, **kwargs): + rel_path = self._construct_path(dataset_id, **kwargs) + extra_dir = kwargs.get('extra_dir', None) + try: + # For the case of extra_files, because we don't have a reference to + # individual files/keys we need to remove the entire directory structure + # with all the files in it. This is easy for the local file system, + # but requires iterating through each individual key in S3 and deleing it. + if entire_dir and extra_dir: + shutil.rmtree(self._get_cache_path(rel_path)) + rs = self.bucket.get_all_keys(prefix=rel_path) + for key in rs: + log.debug("Deleting key %s" % key.name) + key.delete() + return True + else: + # Delete from cache first + os.unlink(self._get_cache_path(rel_path)) + # Delete from S3 as well + if self._key_exists(rel_path): + key = Key(self.bucket, rel_path) + log.debug("Deleting key %s" % key.name) + key.delete() + return True + except S3ResponseError, ex: + log.error("Could not delete key '%s' from S3: %s" % (rel_path, ex)) + except OSError, ex: + log.error('%s delete error %s' % (self._get_filename(dataset_id, **kwargs), ex)) + return False + + def get_data(self, dataset_id, start=0, count=-1, **kwargs): + rel_path = self._construct_path(dataset_id, **kwargs) + # Check cache first and get file if not there + if not self._in_cache(rel_path): + self._pull_into_cache(rel_path) + else: + print "(cccc) Getting '%s' from cache" % self._get_cache_path(rel_path) + # Read the file content from cache + data_file = open(self._get_cache_path(rel_path), 'r') + data_file.seek(start) + content = data_file.read(count) + data_file.close() + return content + + def get_filename(self, dataset_id, **kwargs): + print "S3 get_filename for dataset: %s" % dataset_id + dir_only = kwargs.get('dir_only', False) + rel_path = self._construct_path(dataset_id, **kwargs) + cache_path = self._get_cache_path(rel_path) + # S3 does not recognize directories as files so cannot check if those exist. + # So, if checking dir only, ensure given dir exists in cache and return + # the expected cache path. + # dir_only = kwargs.get('dir_only', False) + # if dir_only: + # if not os.path.exists(cache_path): + # os.makedirs(cache_path) + # return cache_path + # Check if the file exists in the cache first + if self._in_cache(rel_path): + return cache_path + # Check if the file exists in persistent storage and, if it does, pull it into cache + elif self.exists(dataset_id, **kwargs): + if dir_only: # Directories do not get pulled into cache + return cache_path + else: + if self._pull_into_cache(rel_path): + return cache_path + # For the case of retrieving a directory only, return the expected path + # even if it does not exist. + # if dir_only: + # return cache_path + raise ObjectNotFound() + # return cache_path # Until the upload tool does not explicitly create the dataset, return expected path + + def update_from_file(self, dataset_id, file_name=None, create=False, **kwargs): + if create: + self.create(dataset_id, **kwargs) + if self.exists(dataset_id, **kwargs): + rel_path = self._construct_path(dataset_id, **kwargs) + # Chose whether to use the dataset file itself or an alternate file + if file_name: + source_file = os.path.abspath(file_name) + # Copy into cache + cache_file = self._get_cache_path(rel_path) + try: + # FIXME? Should this be a `move`? + shutil.copy2(source_file, cache_file) + self._fix_permissions(cache_file) + except OSError, ex: + log.error("Trouble copying source file '%s' to cache '%s': %s" % (source_file, cache_file, ex)) + else: + source_file = self._get_cache_path(rel_path) + # Update the file on S3 + self._push_to_s3(rel_path, source_file) + else: + raise ObjectNotFound() + + def get_object_url(self, dataset_id, **kwargs): + if self.exists(dataset_id, **kwargs): + rel_path = self._construct_path(dataset_id, **kwargs) + try: + key = Key(self.bucket, rel_path) + return key.generate_url(expires_in = 86400) # 24hrs + except S3ResponseError, ex: + log.warning("Trouble generating URL for dataset '%s': %s" % (rel_path, ex)) + return None + + + +class HierarchicalObjectStore(ObjectStore): + """ + ObjectStore that defers to a list of backends, for getting objects the + first store where the object exists is used, objects are always created + in the first store. + """ + + def __init__(self, backends=[]): + super(HierarchicalObjectStore, self).__init__() + + +def build_object_store_from_config(app): + """ Depending on the configuration setting, invoke the appropriate object store + """ + store = app.config.object_store + if store == 'disk': + return DiskObjectStore(app=app) + elif store == 's3': + os.environ['AWS_ACCESS_KEY_ID'] = app.config.aws_access_key + os.environ['AWS_SECRET_ACCESS_KEY'] = app.config.aws_secret_key + return S3ObjectStore(app=app) + elif store == 'hierarchical': + return HierarchicalObjectStore() + +def convert_bytes(bytes): + """ A helper function used for pretty printing disk usage """ + if bytes is None: + bytes = 0 + bytes = float(bytes) + + if bytes >= 1099511627776: + terabytes = bytes / 1099511627776 + size = '%.2fTB' % terabytes + elif bytes >= 1073741824: + gigabytes = bytes / 1073741824 + size = '%.2fGB' % gigabytes + elif bytes >= 1048576: + megabytes = bytes / 1048576 + size = '%.2fMB' % megabytes + elif bytes >= 1024: + kilobytes = bytes / 1024 + size = '%.2fKB' % kilobytes + else: + size = '%.2fb' % bytes + return size diff --git a/lib/galaxy/objectstore/s3_multipart_upload.py b/lib/galaxy/objectstore/s3_multipart_upload.py new file mode 100644 index 00000000000..85b6b3e34a1 --- /dev/null +++ b/lib/galaxy/objectstore/s3_multipart_upload.py @@ -0,0 +1,85 @@ +#!/usr/bin/env python +""" +Split large file into multiple pieces for upload to S3. +This parallelizes the task over available cores using multiprocessing. +Code mostly taken form CloudBioLinux. +""" +import os +import glob +import subprocess +import contextlib +import functools +import multiprocessing +from multiprocessing.pool import IMapIterator + +import boto + +def map_wrap(f): + @functools.wraps(f) + def wrapper(*args, **kwargs): + return apply(f, *args, **kwargs) + return wrapper + +def mp_from_ids(mp_id, mp_keyname, mp_bucketname): + """Get the multipart upload from the bucket and multipart IDs. + + This allows us to reconstitute a connection to the upload + from within multiprocessing functions. + """ + conn = boto.connect_s3() + bucket = conn.lookup(mp_bucketname) + mp = boto.s3.multipart.MultiPartUpload(bucket) + mp.key_name = mp_keyname + mp.id = mp_id + return mp + +@map_wrap +def transfer_part(mp_id, mp_keyname, mp_bucketname, i, part): + """Transfer a part of a multipart upload. Designed to be run in parallel. + """ + mp = mp_from_ids(mp_id, mp_keyname, mp_bucketname) + print " Transferring", i, part + with open(part) as t_handle: + mp.upload_part_from_file(t_handle, i+1) + os.remove(part) + +def multipart_upload(bucket, s3_key_name, tarball, mb_size, use_rr=True): + """Upload large files using Amazon's multipart upload functionality. + """ + cores = multiprocessing.cpu_count() + print "Initiating multipart upload using %s cores" % cores + def split_file(in_file, mb_size, split_num=5): + prefix = os.path.join(os.path.dirname(in_file), + "%sS3PART" % (os.path.basename(s3_key_name))) + # Split chunks so they are 5MB < chunk < 250MB + split_size = int(max(min(mb_size / (split_num * 2.0), 250), 5)) + if not os.path.exists("%saa" % prefix): + cl = ["split", "-b%sm" % split_size, in_file, prefix] + subprocess.check_call(cl) + return sorted(glob.glob("%s*" % prefix)) + + mp = bucket.initiate_multipart_upload(s3_key_name, reduced_redundancy=use_rr) + with multimap(cores) as pmap: + for _ in pmap(transfer_part, ((mp.id, mp.key_name, mp.bucket_name, i, part) + for (i, part) in + enumerate(split_file(tarball, mb_size, cores)))): + pass + mp.complete_upload() + +@contextlib.contextmanager +def multimap(cores=None): + """Provide multiprocessing imap like function. + + The context manager handles setting up the pool, worked around interrupt issues + and terminating the pool on completion. + """ + if cores is None: + cores = max(multiprocessing.cpu_count() - 1, 1) + def wrapper(func): + def wrap(self, timeout=None): + return func(self, timeout=timeout if timeout is not None else 1e100) + return wrap + IMapIterator.next = wrapper(IMapIterator.next) + pool = multiprocessing.Pool(cores) + yield pool.imap + pool.terminate() diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index ebbdbed9fcf..31922acd39d 100755 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -1701,23 +1701,38 @@ class Tool: Find extra files in the job working directory and move them into the appropriate dataset's files directory """ + # print "Working in collect_associated_files" for name, hda in output.items(): temp_file_path = os.path.join( job_working_directory, "dataset_%s_files" % ( hda.dataset.id ) ) try: - if len( os.listdir( temp_file_path ) ) > 0: - store_file_path = os.path.join( - os.path.join( self.app.config.file_path, *directory_hash_id( hda.dataset.id ) ), - "dataset_%d_files" % hda.dataset.id ) - shutil.move( temp_file_path, store_file_path ) - # Fix permissions - for basedir, dirs, files in os.walk( store_file_path ): - util.umask_fix_perms( basedir, self.app.config.umask, 0777, self.app.config.gid ) - for file in files: - path = os.path.join( basedir, file ) - # Ignore symlinks - if os.path.islink( path ): - continue - util.umask_fix_perms( path, self.app.config.umask, 0666, self.app.config.gid ) + a_files = os.listdir( temp_file_path ) + if len( a_files ) > 0: + for f in a_files: + # print "------ Instructing ObjectStore to update/create file: %s from %s" \ + # % (hda.dataset.id, os.path.join(temp_file_path, f)) + self.app.object_store.update_from_file(hda.dataset.id, + extra_dir="dataset_%d_files" % hda.dataset.id, + alt_name = f, + file_name = os.path.join(temp_file_path, f), + create = True) + # Clean up after being handled by object store. + # FIXME: If the object (e.g., S3) becomes async, this will + # cause issues so add it to the object store functionality? + # shutil.rmtree(temp_file_path) + + # store_file_path = os.path.join( + # os.path.join( self.app.config.file_path, *directory_hash_id( hda.dataset.id ) ), + # "dataset_%d_files" % hda.dataset.id ) + # shutil.move( temp_file_path, store_file_path ) + # # Fix permissions + # for basedir, dirs, files in os.walk( store_file_path ): + # util.umask_fix_perms( basedir, self.app.config.umask, 0777, self.app.config.gid ) + # for file in files: + # path = os.path.join( basedir, file ) + # # Ignore symlinks + # if os.path.islink( path ): + # continue + # util.umask_fix_perms( path, self.app.config.umask, 0666, self.app.config.gid ) except: continue diff --git a/lib/galaxy/tools/actions/__init__.py b/lib/galaxy/tools/actions/__init__.py index 000b6980716..533db36f3a8 100644 --- a/lib/galaxy/tools/actions/__init__.py +++ b/lib/galaxy/tools/actions/__init__.py @@ -282,7 +282,8 @@ class DefaultToolAction( object ): trans.sa_session.flush() trans.app.security_agent.set_all_dataset_permissions( data.dataset, output_permissions ) # Create an empty file immediately - open( data.file_name, "w" ).close() + # open( data.file_name, "w" ).close() + trans.app.object_store.create( data.id ) # Fix permissions util.umask_fix_perms( data.file_name, trans.app.config.umask, 0666 ) # This may not be neccesary with the new parent/child associations diff --git a/lib/galaxy/tools/actions/upload.py b/lib/galaxy/tools/actions/upload.py index 757f7fba866..0f39a299d0d 100644 --- a/lib/galaxy/tools/actions/upload.py +++ b/lib/galaxy/tools/actions/upload.py @@ -1,4 +1,3 @@ -import os from __init__ import ToolAction from galaxy.tools.actions import upload_common diff --git a/lib/galaxy/tools/actions/upload_common.py b/lib/galaxy/tools/actions/upload_common.py index c6b7bb3f53e..b27b0a70377 100644 --- a/lib/galaxy/tools/actions/upload_common.py +++ b/lib/galaxy/tools/actions/upload_common.py @@ -324,13 +324,17 @@ def create_job( trans, params, tool, json_file_path, data_list, folder=None, ret job.add_output_library_dataset( 'output%i' % i, dataset ) # Create an empty file immediately if not dataset.dataset.external_filename: - open( dataset.file_name, "w" ).close() + trans.app.object_store.create( dataset.id ) + print "---> Upload tool created a folder(?) %s with ID %s? %s" % (dataset.file_name, dataset.id, trans.app.object_store.exists(dataset.id)) + # open( dataset.file_name, "w" ).close() else: for i, dataset in enumerate( data_list ): job.add_output_dataset( 'output%i' % i, dataset ) # Create an empty file immediately if not dataset.dataset.external_filename: - open( dataset.file_name, "w" ).close() + trans.app.object_store.create( dataset.id ) + print "---> Upload tool created a file %s with ID %s? %s" % (dataset.file_name, dataset.id, trans.app.object_store.exists(dataset.id)) + # open( dataset.file_name, "w" ).close() job.state = job.states.NEW trans.sa_session.add( job ) trans.sa_session.flush() diff --git a/lib/galaxy/web/controllers/dataset.py b/lib/galaxy/web/controllers/dataset.py index 44552e3af7b..8bec4f465d5 100644 --- a/lib/galaxy/web/controllers/dataset.py +++ b/lib/galaxy/web/controllers/dataset.py @@ -217,7 +217,7 @@ class DatasetInterface( BaseController, UsesAnnotations, UsesHistory, UsesHistor outfname = data.name[0:150] outfname = ''.join(c in valid_chars and c or '_' for c in outfname) if (params.do_action == None): - params.do_action = 'zip' # default + params.do_action = 'zip' # default msg = util.restore_text( params.get( 'msg', '' ) ) messagetype = params.get( 'messagetype', 'done' ) if not data: @@ -300,8 +300,7 @@ class DatasetInterface( BaseController, UsesAnnotations, UsesHistory, UsesHistor archive.wsgi_headeritems = trans.response.wsgi_headeritems() return archive.stream return trans.show_error_message( msg ) - - + @web.expose def get_metadata_file(self, trans, hda_id, metadata_name): """ Allows the downloading of metadata files associated with datasets (eg. bai index for bam files) """ @@ -316,12 +315,8 @@ class DatasetInterface( BaseController, UsesAnnotations, UsesHistory, UsesHistor trans.response.headers["Content-Type"] = "application/octet-stream" trans.response.headers["Content-Disposition"] = "attachment; filename=Galaxy%s-[%s].%s" % (data.hid, fname, file_ext) return open(data.metadata.get(metadata_name).file_name) - - @web.expose - def display(self, trans, dataset_id=None, preview=False, filename=None, to_ext=None, **kwd): - """Catches the dataset id and displays file contents as directed""" - composite_extensions = trans.app.datatypes_registry.get_composite_extensions( ) - composite_extensions.append('html') # for archiving composite datatypes + + def _check_dataset(self, trans, dataset_id): # DEPRECATION: We still support unencoded ids for backward compatibility try: data = trans.sa_session.query( trans.app.model.HistoryDatasetAssociation ).get( trans.security.decode_id( dataset_id ) ) @@ -340,9 +335,36 @@ class DatasetInterface( BaseController, UsesAnnotations, UsesHistory, UsesHistor if data.state == trans.model.Dataset.states.UPLOAD: return trans.show_error_message( "Please wait until this dataset finishes uploading before attempting to view it." ) + return data + + @web.expose + @web.json + def transfer_status(self, trans, dataset_id, filename=None): + """ Primarily used for the S3ObjectStore - get the status of data transfer + if the file is not in cache """ + data = self._check_dataset(trans, dataset_id) + print "dataset.py -> transfer_status: Checking transfer status for dataset %s..." % data.id + + # Pulling files in extra_files_path into cache is not handled via this + # method but that's primarily because those files are typically linked to + # through tool's output page anyhow so tying a JavaScript event that will + # call this method does not seem doable? + if trans.app.object_store.file_ready(data.id): + return True + else: + return False + + @web.expose + def display(self, trans, dataset_id=None, preview=False, filename=None, to_ext=None, **kwd): + """Catches the dataset id and displays file contents as directed""" + composite_extensions = trans.app.datatypes_registry.get_composite_extensions( ) + composite_extensions.append('html') # for archiving composite datatypes + data = self._check_dataset(trans, dataset_id) + if filename and filename != "index": # For files in extra_files_path - file_path = os.path.join( data.extra_files_path, filename ) + file_path = os.path.join( data.extra_files_path, filename ) # remove after debugging + file_path = trans.app.object_store.get_filename(data.id, extra_dir='dataset_%s_files' % data.id, alt_name=filename) if os.path.exists( file_path ): if os.path.isdir( file_path ): return trans.show_error_message( "Directory listing is not allowed." ) #TODO: Reconsider allowing listing of directories? diff --git a/lib/galaxy/web/controllers/history.py b/lib/galaxy/web/controllers/history.py index 15c4032539a..5b236a0632d 100644 --- a/lib/galaxy/web/controllers/history.py +++ b/lib/galaxy/web/controllers/history.py @@ -581,7 +581,7 @@ class HistoryController( BaseController, Sharable, UsesAnnotations, UsesItemRati trans.response.set_content_type( 'application/x-gzip' ) else: trans.response.set_content_type( 'application/x-tar' ) - return open( jeha.dataset.file_name ) + return trans.app.object_store.get_data(jeha.dataset.id) elif jeha.job.state in [ model.Job.states.RUNNING, model.Job.states.QUEUED, model.Job.states.WAITING ]: return trans.show_message( "Still exporting history %(n)s; please check back soon. Link: %(s)s" \ % ( { 'n' : history.name, 's' : url_for( action="export_archive", id=id, qualified=True ) } ) ) diff --git a/templates/dataset/display.mako b/templates/dataset/display.mako index c1b925638b7..4085fadbc2d 100755 --- a/templates/dataset/display.mako +++ b/templates/dataset/display.mako @@ -9,11 +9,11 @@ <%def name="init()"> <% - self.has_left_panel=False - self.has_right_panel=True - self.message_box_visible=False - self.active_view="user" - self.overlay_visible=False + self.has_left_panel=False + self.has_right_panel=True + self.message_box_visible=False + self.active_view="user" + self.overlay_visible=False %> @@ -44,10 +44,10 @@ <%def name="center_panel()">
-
- ${get_class_display_name( item.__class__ )} - | ${get_item_name( item ) | h} -
+
+ ${get_class_display_name( item.__class__ )} + | ${get_item_name( item ) | h} +
diff --git a/templates/root/history.mako b/templates/root/history.mako index 13962bd6488..1f35a1954ab 100755 --- a/templates/root/history.mako +++ b/templates/root/history.mako @@ -140,6 +140,34 @@ $(function() { return false; }); }); + + // Check to see if the dataset data is cached or needs to be pulled in + // via objectstore + $(this).find("a.display").each( function() { + var history_item = $(this).parents(".historyItem")[0]; + var history_id = history_item.id.split( "-" )[1]; + $(this).click(function() { + check_transfer_status($(this), history_id); + }); + }); + + // If dataset data is not cached, keep making ajax calls to check on the + // data status and update the dataset UI element accordingly + function check_transfer_status(link, history_id) { + $.getJSON("${h.url_for( controller='dataset', action='transfer_status', dataset_id='XXX' )}".replace( 'XXX', link.attr("dataset_id") ), + function(ready) { + if (ready === false) { + // $("
").text("Data is loading from S3... please be patient").appendTo(link.parent()); + $( '#historyItem-' + history_id).removeClass( "historyItem-ok" ); + $( '#historyItem-' + history_id).addClass( "historyItem-running" ); + setTimeout(function(){check_transfer_status(link, history_id)}, 1000); + } else { + $( '#historyItem-' + history_id).removeClass( "historyItem-running" ); + $( '#historyItem-' + history_id).addClass( "historyItem-ok" ); + } + } + ); + } // Undelete link $(this).find("a.historyItemUndelete").each( function() { diff --git a/templates/root/history_common.mako b/templates/root/history_common.mako index fc79712e098..c4af0948e15 100755 --- a/templates/root/history_common.mako +++ b/templates/root/history_common.mako @@ -98,7 +98,7 @@ %if data.purged: %else: - Date: Thu, 21 Jul 2011 10:44:27 -0400 Subject: [PATCH 006/362] Added config options to universe_wsgi.ini.sample --- universe_wsgi.ini.sample | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/universe_wsgi.ini.sample b/universe_wsgi.ini.sample index c136c0879e1..473711d2457 100644 --- a/universe_wsgi.ini.sample +++ b/universe_wsgi.ini.sample @@ -414,6 +414,17 @@ use_interactive = True # -- Beta features +# Object store mode (valid options are: disk, s3, hierarchical) +#object_store = s3 +#aws_access_key = +#aws_secret_key = +#s3_bucket = +#use_reduced_redundancy = True +# Size (in GB) that the cache used by object store should be limited to. +# If the value is not specified, the cache size will be limited only by the file +# system size. +#object_store_cache_size = 100 + # Enable Galaxy to communicate directly with a sequencer #enable_sequencer_communication = False From dac2bbf1f5706c74d3666fec8f0d79283e37a6fa Mon Sep 17 00:00:00 2001 From: Enis Afgan Date: Mon, 25 Jul 2011 14:19:51 -0400 Subject: [PATCH 007/362] Fix for permission setting on files downloaded from S3 into cache --- lib/galaxy/objectstore/__init__.py | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/lib/galaxy/objectstore/__init__.py b/lib/galaxy/objectstore/__init__.py index 076ca8cfbd1..cbf3bbd83e3 100644 --- a/lib/galaxy/objectstore/__init__.py +++ b/lib/galaxy/objectstore/__init__.py @@ -442,12 +442,12 @@ class S3ObjectStore(ObjectStore): def _fix_permissions(self, rel_path): """ Set permissions on rel_path""" - for basedir, dirs, files in os.walk( rel_path ): - util.umask_fix_perms( basedir, self.app.config.umask, 0777, self.app.config.gid ) + for basedir, dirs, files in os.walk(rel_path): + util.umask_fix_perms(basedir, self.app.config.umask, 0777, self.app.config.gid) for f in files: - path = os.path.join( basedir, f ) + path = os.path.join(basedir, f) # Ignore symlinks - if os.path.islink( path ): + if os.path.islink(path): continue util.umask_fix_perms( path, self.app.config.umask, 0666, self.app.config.gid ) @@ -545,7 +545,7 @@ class S3ObjectStore(ObjectStore): os.makedirs(self._get_cache_path(rel_path_dir)) # Now pull in the file ok = self._download(rel_path) - self._fix_permissions(rel_path) + self._fix_permissions(self._get_cache_path(rel_path_dir)) return ok def _transfer_cb(self, complete, total): @@ -569,14 +569,14 @@ class S3ObjectStore(ObjectStore): if ret_code == 127: self.transfer_progress = 0 # Reset transfer progress counter key.get_contents_to_filename(self._get_cache_path(rel_path), cb=self._transfer_cb, num_cb=10) - print "(ssss) Pulled key '%s' into cache to %s" % (rel_path, self._get_cache_path(rel_path)) + print "(ssss1) Pulled key '%s' into cache to %s" % (rel_path, self._get_cache_path(rel_path)) return True else: ncores = multiprocessing.cpu_count() url = key.generate_url(7200) ret_code = subprocess.call("axel -a -n %s '%s'" % (ncores, url)) if ret_code == 0: - print "(ssss) Parallel pulled key '%s' into cache to %s" % (rel_path, self._get_cache_path(rel_path)) + print "(ssss2) Parallel pulled key '%s' into cache to %s" % (rel_path, self._get_cache_path(rel_path)) return True except S3ResponseError, ex: log.error("Problem downloading key '%s' from S3 bucket '%s': %s" % (rel_path, self.bucket.name, ex)) From ab6e0d30b1b288f0f3a35516ea663954bacbba31 Mon Sep 17 00:00:00 2001 From: Nate Coraor Date: Wed, 27 Jul 2011 10:48:03 -0400 Subject: [PATCH 008/362] Closed the feature/ws branch, changes from this branch were merged in 5827:f3a1086fac91. From 29253409ac6a83b3a2da17342419233c66dd78b3 Mon Sep 17 00:00:00 2001 From: Enis Afgan Date: Mon, 1 Aug 2011 17:35:26 -0400 Subject: [PATCH 009/362] Metadata is now being handled by the ObjectStore as well --- lib/galaxy/app.py | 3 +- lib/galaxy/datatypes/metadata.py | 6 ++- lib/galaxy/jobs/__init__.py | 2 +- lib/galaxy/jobs/runners/local.py | 2 +- lib/galaxy/model/__init__.py | 11 +----- lib/galaxy/objectstore/__init__.py | 48 ++++++++++++----------- lib/galaxy/tools/actions/upload_common.py | 2 - scripts/set_metadata.py | 30 +++++++++++++- 8 files changed, 64 insertions(+), 40 deletions(-) diff --git a/lib/galaxy/app.py b/lib/galaxy/app.py index 63c65059d99..fa5e5e0ad18 100644 --- a/lib/galaxy/app.py +++ b/lib/galaxy/app.py @@ -32,7 +32,7 @@ class UniverseApplication( object ): from galaxy.model.migrate.check import create_or_verify_database create_or_verify_database( db_url, kwargs.get( 'global_conf', {} ).get( '__file__', None ), self.config.database_engine_options ) # Object store manager - self.object_store = build_object_store_from_config(self) + self.object_store = build_object_store_from_config(self.config) # Setup the database engine and ORM from galaxy.model import mapping self.model = mapping.init( self.config.file_path, @@ -94,5 +94,6 @@ class UniverseApplication( object ): def shutdown( self ): self.job_manager.shutdown() + self.object_store.shutdown() if self.heartbeat: self.heartbeat.shutdown() diff --git a/lib/galaxy/datatypes/metadata.py b/lib/galaxy/datatypes/metadata.py index 174aae0c5c4..f4713d8f1de 100644 --- a/lib/galaxy/datatypes/metadata.py +++ b/lib/galaxy/datatypes/metadata.py @@ -411,6 +411,7 @@ class FileParameter( MetadataParameter ): mf = galaxy.model.MetadataFile() mf.id = value #we assume this is a valid id, since we cannot check it return mf + def make_copy( self, value, target_context, source_context ): value = self.wrap( value ) if value: @@ -438,8 +439,11 @@ class FileParameter( MetadataParameter ): if mf is None: mf = self.new_file( dataset = parent, **value.kwds ) shutil.move( value.file_name, mf.file_name ) + # Ensure the metadata file gets updated with content + parent.dataset.object_store.update_from_file( parent.dataset.id, file_name=mf.file_name, extra_dir='_metadata_files', extra_dir_at_root=True, alt_name=os.path.basename(mf.file_name) ) value = mf.id return value + def to_external_value( self, value ): """ Turns a value read from a metadata into its value to be pushed directly into the external dict. @@ -461,7 +465,7 @@ class FileParameter( MetadataParameter ): #we will be copying its contents into the MetadataFile objects filename after restoring from JSON #we do not include 'dataset' in the kwds passed, as from_JSON_value() will handle this for us return MetadataTempFile( **kwds ) - + #This class is used when a database file connection is not available class MetadataTempFile( object ): tmp_dir = 'database/tmp' #this should be overwritten as necessary in calling scripts diff --git a/lib/galaxy/jobs/__init__.py b/lib/galaxy/jobs/__init__.py index 00cd5026165..6dac9cb54d5 100644 --- a/lib/galaxy/jobs/__init__.py +++ b/lib/galaxy/jobs/__init__.py @@ -560,7 +560,7 @@ class JobWrapper( object ): dataset.set_size() # Update (non-library) job output datasets through the object store if dataset not in job.output_library_datasets: - print "===+=== Handing dataset '%s' with name '%s' to object store" % (dataset.id, dataset.file_name) + print "===++=== Job finish method handing dataset '%s' to object store" % dataset.file_name self.app.object_store.update_from_file(dataset.id, create=True) if context['stderr']: dataset.blurb = "error" diff --git a/lib/galaxy/jobs/runners/local.py b/lib/galaxy/jobs/runners/local.py index e5357a27ed2..ff509dab5fe 100644 --- a/lib/galaxy/jobs/runners/local.py +++ b/lib/galaxy/jobs/runners/local.py @@ -118,7 +118,7 @@ class LocalJobRunner( BaseJobRunner ): preexec_fn = os.setpgrp ) job_wrapper.external_output_metadata.set_job_runner_external_pid( external_metadata_proc.pid, self.sa_session ) external_metadata_proc.wait() - log.debug( 'execution of external set_meta finished for job %d' % job_wrapper.job_id ) + log.debug( 'execution of external set_meta for job %d finished' % job_wrapper.job_id ) # Finish the job try: diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 493f64ce91e..46e55589422 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -533,13 +533,12 @@ class Dataset( object ): self.external_filename = external_filename self._extra_files_path = extra_files_path self.file_size = file_size + def get_file_name( self ): if not self.external_filename: assert self.id is not None, "ID must be set before filename used (commit the object)" assert self.object_store is not None, "Object Store has not been initialized for dataset %s" % self.id - print "Calling get_filename 1", self.object_store filename = self.object_store.get_filename( self.id ) - # print 'getting filename: ', filename if not self.object_store.exists( self.id ): # Create directory if it does not exist self.object_store.create( self.id, dir_only=True ) @@ -556,7 +555,6 @@ class Dataset( object ): file_name = property( get_file_name, set_file_name ) @property def extra_files_path( self ): - print "Calling get_filename 2", self.object_store return self.object_store.get_filename( self.id, dir_only=True, extra_dir=self._extra_files_path or "dataset_%d_files" % self.id) def get_size( self, nice_size=False ): """Returns the size of the data on disk""" @@ -1583,15 +1581,8 @@ class MetadataFile( object ): assert self.id is not None, "ID must be set before filename used (commit the object)" # Ensure the directory structure and the metadata file object exist try: - # self.history_dataset - # print "Dataset.file_path: %s, self.id: %s, self.history_dataset.dataset.object_store: %s" \ - # % (Dataset.file_path, self.id, self.history_dataset.dataset.object_store) self.history_dataset.dataset.object_store.create( self.id, extra_dir='_metadata_files', extra_dir_at_root=True, alt_name="metadata_%d.dat" % self.id ) - print "Calling get_filename 3", self.object_store path = self.history_dataset.dataset.object_store.get_filename( self.id, extra_dir='_metadata_files', extra_dir_at_root=True, alt_name="metadata_%d.dat" % self.id ) - print "Created metadata file at path: %s" % path - self.library_dataset - # raise return path except AttributeError: # In case we're not working with the history_dataset diff --git a/lib/galaxy/objectstore/__init__.py b/lib/galaxy/objectstore/__init__.py index cbf3bbd83e3..d318a1f4b5b 100644 --- a/lib/galaxy/objectstore/__init__.py +++ b/lib/galaxy/objectstore/__init__.py @@ -179,9 +179,9 @@ class DiskObjectStore(ObjectStore): Standard Galaxy object store, stores objects in files under a specific directory on disk. """ - def __init__(self, app): + def __init__(self, config): super(DiskObjectStore, self).__init__() - self.file_path = app.config.file_path + self.file_path = config.file_path def _get_filename(self, dataset_id, dir_only=False, extra_dir=None, extra_dir_at_root=False, alt_name=None): """Class method that returns the absolute path for the file corresponding @@ -344,14 +344,14 @@ class S3ObjectStore(ObjectStore): cache exists that is used as an intermediate location for files between Galaxy and S3. """ - def __init__(self, app): + def __init__(self, config): super(S3ObjectStore, self).__init__() - self.app = app - self.staging_path = self.app.config.file_path + self.config = config + self.staging_path = self.config.file_path self.s3_conn = S3Connection() - self.bucket = self._get_bucket(self.app.config.s3_bucket) - self.use_rr = self.app.config.use_reduced_redundancy - self.cache_size = self.app.config.object_store_cache_size * 1073741824 # Convert GBs to bytes + self.bucket = self._get_bucket(self.config.s3_bucket) + self.use_rr = self.config.use_reduced_redundancy + self.cache_size = self.config.object_store_cache_size * 1073741824 # Convert GBs to bytes self.transfer_progress = 0 # Clean cache only if value is set in universe_wsgi.ini if self.cache_size != -1: @@ -443,13 +443,13 @@ class S3ObjectStore(ObjectStore): def _fix_permissions(self, rel_path): """ Set permissions on rel_path""" for basedir, dirs, files in os.walk(rel_path): - util.umask_fix_perms(basedir, self.app.config.umask, 0777, self.app.config.gid) + util.umask_fix_perms(basedir, self.config.umask, 0777, self.config.gid) for f in files: path = os.path.join(basedir, f) # Ignore symlinks if os.path.islink(path): continue - util.umask_fix_perms( path, self.app.config.umask, 0666, self.app.config.gid ) + util.umask_fix_perms( path, self.config.umask, 0666, self.config.gid ) def _construct_path(self, dataset_id, dir_only=None, extra_dir=None, extra_dir_at_root=False, alt_name=None): rel_path = os.path.join(*directory_hash_id(dataset_id)) @@ -594,12 +594,16 @@ class S3ObjectStore(ObjectStore): source_file = source_file if source_file else self._get_cache_path(rel_path) if os.path.exists(source_file): key = Key(self.bucket, rel_path) + if os.path.getsize(source_file) == 0 and key.exists(): + log.debug("Wanted to push file '%s' to S3 key '%s' but its size is 0; skipping." % (source_file, rel_path)) + return True if from_string: key.set_contents_from_string(from_string, reduced_redundancy=self.use_rr) log.debug("Pushed data from string '%s' to key '%s'" % (from_string, rel_path)) else: start_time = datetime.now() - print "[%s] Pushing cache file '%s' to key '%s'" % (start_time, source_file, rel_path) + # print "Pushing cache file '%s' of size %s bytes to key '%s'" % (source_file, os.path.getsize(source_file), rel_path) + # print "+ Push started at '%s'" % start_time mb_size = os.path.getsize(source_file) / 1e6 if mb_size < 60: self.transfer_progress = 0 # Reset transfer progress counter @@ -607,10 +611,9 @@ class S3ObjectStore(ObjectStore): cb=self._transfer_cb, num_cb=10) else: multipart_upload(self.bucket, key.name, source_file, mb_size, use_rr=self.use_rr) - # self._multipart_upload(key.name, source_file, mb_size) end_time = datetime.now() - print "Push ended at '%s'; it lasted '%s'" % (end_time, end_time-start_time) - log.debug("Pushed cache file '%s' to key '%s'" % (source_file, rel_path)) + # print "+ Push ended at '%s'; %s bytes transfered in %ssec" % (end_time, os.path.getsize(source_file), end_time-start_time) + log.debug("Pushed cache file '%s' to key '%s' (%s bytes transfered in %s sec)" % (source_file, rel_path, os.path.getsize(source_file), end_time-start_time)) return True else: log.error("Tried updating key '%s' from source file '%s', but source file does not exist." @@ -788,8 +791,9 @@ class S3ObjectStore(ObjectStore): # Copy into cache cache_file = self._get_cache_path(rel_path) try: - # FIXME? Should this be a `move`? - shutil.copy2(source_file, cache_file) + if source_file != cache_file: + # FIXME? Should this be a `move`? + shutil.copy2(source_file, cache_file) self._fix_permissions(cache_file) except OSError, ex: log.error("Trouble copying source file '%s' to cache '%s': %s" % (source_file, cache_file, ex)) @@ -823,16 +827,16 @@ class HierarchicalObjectStore(ObjectStore): super(HierarchicalObjectStore, self).__init__() -def build_object_store_from_config(app): +def build_object_store_from_config(config): """ Depending on the configuration setting, invoke the appropriate object store """ - store = app.config.object_store + store = config.object_store if store == 'disk': - return DiskObjectStore(app=app) + return DiskObjectStore(config=config) elif store == 's3': - os.environ['AWS_ACCESS_KEY_ID'] = app.config.aws_access_key - os.environ['AWS_SECRET_ACCESS_KEY'] = app.config.aws_secret_key - return S3ObjectStore(app=app) + os.environ['AWS_ACCESS_KEY_ID'] = config.aws_access_key + os.environ['AWS_SECRET_ACCESS_KEY'] = config.aws_secret_key + return S3ObjectStore(config=config) elif store == 'hierarchical': return HierarchicalObjectStore() diff --git a/lib/galaxy/tools/actions/upload_common.py b/lib/galaxy/tools/actions/upload_common.py index b27b0a70377..b0517e13c60 100644 --- a/lib/galaxy/tools/actions/upload_common.py +++ b/lib/galaxy/tools/actions/upload_common.py @@ -325,7 +325,6 @@ def create_job( trans, params, tool, json_file_path, data_list, folder=None, ret # Create an empty file immediately if not dataset.dataset.external_filename: trans.app.object_store.create( dataset.id ) - print "---> Upload tool created a folder(?) %s with ID %s? %s" % (dataset.file_name, dataset.id, trans.app.object_store.exists(dataset.id)) # open( dataset.file_name, "w" ).close() else: for i, dataset in enumerate( data_list ): @@ -333,7 +332,6 @@ def create_job( trans, params, tool, json_file_path, data_list, folder=None, ret # Create an empty file immediately if not dataset.dataset.external_filename: trans.app.object_store.create( dataset.id ) - print "---> Upload tool created a file %s with ID %s? %s" % (dataset.file_name, dataset.id, trans.app.object_store.exists(dataset.id)) # open( dataset.file_name, "w" ).close() job.state = job.states.NEW trans.sa_session.add( job ) diff --git a/scripts/set_metadata.py b/scripts/set_metadata.py index 792f3c18b71..a972ed05cfe 100644 --- a/scripts/set_metadata.py +++ b/scripts/set_metadata.py @@ -27,6 +27,9 @@ galaxy.datatypes.metadata.DATABASE_CONNECTION_AVAILABLE = False #Let metadata kn from galaxy.util import stringify_dictionary_keys from galaxy.util.json import from_json_string from sqlalchemy.orm import clear_mappers +from galaxy.objectstore import build_object_store_from_config +from galaxy import config +import ConfigParser def __main__(): file_path = sys.argv.pop( 1 ) @@ -34,11 +37,32 @@ def __main__(): galaxy.model.Dataset.file_path = file_path galaxy.datatypes.metadata.MetadataTempFile.tmp_dir = tmp_dir + # Set up reference to object store + # First, read in the main config file for Galaxy; this is required because + # the object store configuration is stored there + conf = ConfigParser.ConfigParser() + config_file_name = 'universe_wsgi.ini' # Safe assumption? + conf.read(config_file_name) + conf_dict = {} + for section in conf.sections(): + for option in conf.options(section): + try: + conf_dict[option] = conf.get(section, option) + except ConfigParser.InterpolationMissingOptionError: + # Because this is not called from Paste Script, %(here)s variable + # is not initialized in the config file so skip those fields - + # just need not to use any such fields for the object store conf... + log.debug("Did not load option %s from %s" % (option, config_file_name)) + # config object is required by ObjectStore class so create it now + universe_config = config.Configuration(**conf_dict) + object_store = build_object_store_from_config(universe_config) + galaxy.model.Dataset.object_store = object_store + # Set up datatypes registry config_root = sys.argv.pop( 1 ) datatypes_config = sys.argv.pop( 1 ) galaxy.model.set_datatypes_registry( galaxy.datatypes.registry.Registry( config_root, datatypes_config ) ) - + job_metadata = sys.argv.pop( 1 ) ext_override = dict() if job_metadata != "None" and os.path.exists( job_metadata ): @@ -83,5 +107,7 @@ def __main__(): except Exception, e: simplejson.dump( ( False, str( e ) ), open( filename_results_code, 'wb+' ) ) #setting metadata has failed somehow clear_mappers() + # Shut down any additional threads that might have been created via the ObjectStore + object_store.shutdown() -__main__() +__main__() \ No newline at end of file From cdb1b47d8f17b233c514482917d624fc57a637fa Mon Sep 17 00:00:00 2001 From: John Duddy Date: Thu, 18 Aug 2011 14:16:03 -0700 Subject: [PATCH 010/362] Allow 2 new optional parameters to workflow/run controller method: history_id: an encoded history id to use. Will not permantently switch user's current id hide_fixed_params: Initially hides all workflow parameters that are not "Set at Runtime" and all workflow steps that only contain them. Intended to reduce clutter when launching "canned" workflows. Also added configurable feature that governs how initial values are selected from the history for workflow runtime input. When enabled, this feature causes Galaxy to use each input only once until it has used them all. This is for the paired-end scenario, to attempt to match inputs correctly by default. --- lib/galaxy/config.py | 1 + lib/galaxy/tools/parameters/basic.py | 31 ++- lib/galaxy/web/controllers/workflow.py | 322 +++++++++++++------------ templates/workflow/run.mako | 102 ++++---- universe_wsgi.ini.sample | 17 ++ 5 files changed, 275 insertions(+), 198 deletions(-) diff --git a/lib/galaxy/config.py b/lib/galaxy/config.py index 2f3c5fa7a7a..4b3c57d3484 100644 --- a/lib/galaxy/config.py +++ b/lib/galaxy/config.py @@ -46,6 +46,7 @@ class Configuration( object ): self.enable_api = string_as_bool( kwargs.get( 'enable_api', False ) ) self.enable_openid = string_as_bool( kwargs.get( 'enable_openid', False ) ) self.enable_quotas = string_as_bool( kwargs.get( 'enable_quotas', False ) ) + self.enable_unique_workflow_defaults = string_as_bool ( kwargs.get ('enable_unique_workflow_defaults', False ) ) self.tool_path = resolve_path( kwargs.get( "tool_path", "tools" ), self.root ) self.tool_data_path = resolve_path( kwargs.get( "tool_data_path", "tool-data" ), os.getcwd() ) self.len_file_path = kwargs.get( "len_file_path", resolve_path(os.path.join(self.tool_data_path, 'shared','ucsc','chrom'), self.root) ) diff --git a/lib/galaxy/tools/parameters/basic.py b/lib/galaxy/tools/parameters/basic.py index c02a0346edb..fd1eeba3608 100644 --- a/lib/galaxy/tools/parameters/basic.py +++ b/lib/galaxy/tools/parameters/basic.py @@ -75,6 +75,16 @@ class ToolParameter( object ): """ return None + def get_initial_value_from_history_prevent_repeats( self, trans, context, already_used ): + """ + Get the starting value for the parameter, but if fetching from the history, try + to find a value that has not yet been used. already_used is a list of objects that + tools must manipulate (by adding to it) to store a memento that they can use to detect + if a value has already been chosen from the history. This is to support the capability to + choose each dataset once + """ + return self.get_initial_value(trans, context); + def get_required_enctype( self ): """ If this parameter needs the form to have a specific encoding @@ -1385,6 +1395,9 @@ class DataToolParameter( ToolParameter ): return field def get_initial_value( self, trans, context ): + return self.get_initial_value_from_history_prevent_repeats(trans, context, None); + + def get_initial_value_from_history_prevent_repeats( self, trans, context, already_used ): """ NOTE: This is wasteful since dynamic options and dataset collection happens twice (here and when generating HTML). @@ -1397,7 +1410,7 @@ class DataToolParameter( ToolParameter ): assert history is not None, "DataToolParameter requires a history" if self.optional: return None - most_recent_dataset = [None] + most_recent_dataset = [] filter_value = None if self.options: try: @@ -1423,15 +1436,19 @@ class DataToolParameter( ToolParameter ): data = converted_dataset if not is_valid or ( self.options and self._options_filter_attribute( data ) != filter_value ): continue - most_recent_dataset[0] = data + most_recent_dataset.append(data) # Also collect children via association object dataset_collector( data.children ) dataset_collector( history.datasets ) - most_recent_dataset = most_recent_dataset.pop() - if most_recent_dataset is not None: - return most_recent_dataset - else: - return '' + most_recent_dataset.reverse() + if already_used is not None: + for val in most_recent_dataset: + if val is not None and val not in already_used: + already_used.append(val) + return val + if len(most_recent_dataset) > 0: + return most_recent_dataset[0] + return '' def from_html( self, value, trans, other_values={} ): # Can't look at history in workflow mode, skip validation and such, diff --git a/lib/galaxy/web/controllers/workflow.py b/lib/galaxy/web/controllers/workflow.py index 497d104e806..83516f0e305 100644 --- a/lib/galaxy/web/controllers/workflow.py +++ b/lib/galaxy/web/controllers/workflow.py @@ -1258,7 +1258,7 @@ class WorkflowController( BaseController, Sharable, UsesStoredWorkflow, UsesAnno ## % ( workflow_name, web.url_for( action='editor', id=trans.security.encode_id(stored.id) ) ) ) @web.expose - def run( self, trans, id, **kwargs ): + def run( self, trans, id, history_id=None, hide_fixed_params=False, **kwargs ): stored = self.get_stored_workflow( trans, id, check_ownership=False ) user = trans.get_user() if stored.user != user: @@ -1279,163 +1279,185 @@ class WorkflowController( BaseController, Sharable, UsesStoredWorkflow, UsesAnno errors = {} has_upgrade_messages = False has_errors = False - if kwargs: - # If kwargs were provided, the states for each step should have - # been POSTed - # Get the kwarg keys for data inputs - input_keys = filter(lambda a: a.endswith('|input'), kwargs) - # Example: prefixed='2|input' - # Check if one of them is a list - multiple_input_key = None - multiple_inputs = [None] - for input_key in input_keys: - if isinstance(kwargs[input_key], list): - multiple_input_key = input_key - multiple_inputs = kwargs[input_key] - # List to gather values for the template - invocations=[] - for input_number, single_input in enumerate(multiple_inputs): - # Example: single_input='1', single_input='2', etc... - # 'Fix' the kwargs, to have only the input for this iteration - if multiple_input_key: - kwargs[multiple_input_key] = single_input + saved_history = None + if history_id is not None: + saved_history = trans.get_history(); + try: + decoded_history_id = trans.security.decode_id( history_id ) + history = trans.sa_session.query(trans.app.model.History).get(decoded_history_id) + if history.user != trans.user and not trans.user_is_admin(): + if trans.sa_session.query(trans.app.model.HistoryUserShareAssociation).filter_by(user=trans.user, history=history).count() == 0: + error("History is not owned by or shared with current user") + trans.set_history(history) + except TypeError: + error("Malformed history id ( %s ) specified, unable to decode." % str( history_id )) + except: + error("That history does not exist.") + try: # use a try/finally block to restore the user's current history + if kwargs: + # If kwargs were provided, the states for each step should have + # been POSTed + # Get the kwarg keys for data inputs + input_keys = filter(lambda a: a.endswith('|input'), kwargs) + # Example: prefixed='2|input' + # Check if one of them is a list + multiple_input_key = None + multiple_inputs = [None] + for input_key in input_keys: + if isinstance(kwargs[input_key], list): + multiple_input_key = input_key + multiple_inputs = kwargs[input_key] + # List to gather values for the template + invocations=[] + for input_number, single_input in enumerate(multiple_inputs): + # Example: single_input='1', single_input='2', etc... + # 'Fix' the kwargs, to have only the input for this iteration + if multiple_input_key: + kwargs[multiple_input_key] = single_input + for step in workflow.steps: + step.upgrade_messages = {} + # Connections by input name + step.input_connections_by_name = \ + dict( ( conn.input_name, conn ) for conn in step.input_connections ) + # Extract just the arguments for this step by prefix + p = "%s|" % step.id + l = len(p) + step_args = dict( ( k[l:], v ) for ( k, v ) in kwargs.iteritems() if k.startswith( p ) ) + step_errors = None + if step.type == 'tool' or step.type is None: + module = module_factory.from_workflow_step( trans, step ) + # Fix any missing parameters + step.upgrade_messages = module.check_and_update_state() + if step.upgrade_messages: + has_upgrade_messages = True + # Any connected input needs to have value DummyDataset (these + # are not persisted so we need to do it every time) + module.add_dummy_datasets( connections=step.input_connections ) + # Get the tool + tool = module.tool + # Get the state + step.state = state = module.state + # Get old errors + old_errors = state.inputs.pop( "__errors__", {} ) + # Update the state + step_errors = tool.update_state( trans, tool.inputs, step.state.inputs, step_args, + update_only=True, old_errors=old_errors ) + else: + # Fix this for multiple inputs + module = step.module = module_factory.from_workflow_step( trans, step ) + state = step.state = module.decode_runtime_state( trans, step_args.pop( "tool_state" ) ) + step_errors = module.update_runtime_state( trans, state, step_args ) + if step_errors: + errors[step.id] = state.inputs["__errors__"] = step_errors + if 'run_workflow' in kwargs and not errors: + new_history = None + if 'new_history' in kwargs: + if 'new_history_name' in kwargs and kwargs['new_history_name'] != '': + nh_name = kwargs['new_history_name'] + else: + nh_name = "History from %s workflow" % workflow.name + if multiple_input_key: + nh_name = '%s %d' % (nh_name, input_number + 1) + new_history = trans.app.model.History( user=trans.user, name=nh_name ) + trans.sa_session.add( new_history ) + # Run each step, connecting outputs to inputs + workflow_invocation = model.WorkflowInvocation() + workflow_invocation.workflow = workflow + outputs = odict() + for i, step in enumerate( workflow.steps ): + # Execute module + job = None + if step.type == 'tool' or step.type is None: + tool = trans.app.toolbox.tools_by_id[ step.tool_id ] + input_values = step.state.inputs + # Connect up + def callback( input, value, prefixed_name, prefixed_label ): + if isinstance( input, DataToolParameter ): + if prefixed_name in step.input_connections_by_name: + conn = step.input_connections_by_name[ prefixed_name ] + return outputs[ conn.output_step.id ][ conn.output_name ] + visit_input_values( tool.inputs, step.state.inputs, callback ) + # Execute it + job, out_data = tool.execute( trans, step.state.inputs, history=new_history) + outputs[ step.id ] = out_data + # Create new PJA associations with the created job, to be run on completion. + # PJA Parameter Replacement (only applies to immediate actions-- rename specifically, for now) + # Pass along replacement dict with the execution of the PJA so we don't have to modify the object. + replacement_dict = {} + for k, v in kwargs.iteritems(): + if k.startswith('wf_parm|'): + replacement_dict[k[8:]] = v + for pja in step.post_job_actions: + if pja.action_type in ActionBox.immediate_actions: + ActionBox.execute(trans.app, trans.sa_session, pja, job, replacement_dict) + else: + job.add_post_job_action(pja) + else: + job, out_data = step.module.execute( trans, step.state ) + outputs[ step.id ] = out_data + # Record invocation + workflow_invocation_step = model.WorkflowInvocationStep() + workflow_invocation_step.workflow_invocation = workflow_invocation + workflow_invocation_step.workflow_step = step + workflow_invocation_step.job = job + # All jobs ran sucessfully, so we can save now + trans.sa_session.add( workflow_invocation ) + invocations.append({'outputs': outputs, + 'new_history': new_history}) + trans.sa_session.flush() + return trans.fill_template( "workflow/run_complete.mako", + workflow=stored, + invocations=invocations ) + else: + # Prepare each step + missing_tools = [] for step in workflow.steps: step.upgrade_messages = {} - # Connections by input name - step.input_connections_by_name = \ - dict( ( conn.input_name, conn ) for conn in step.input_connections ) - # Extract just the arguments for this step by prefix - p = "%s|" % step.id - l = len(p) - step_args = dict( ( k[l:], v ) for ( k, v ) in kwargs.iteritems() if k.startswith( p ) ) - step_errors = None + # Contruct modules if step.type == 'tool' or step.type is None: - module = module_factory.from_workflow_step( trans, step ) - # Fix any missing parameters - step.upgrade_messages = module.check_and_update_state() + # Restore the tool state for the step + step.module = module_factory.from_workflow_step( trans, step ) + if not step.module: + if step.tool_id not in missing_tools: + missing_tools.append(step.tool_id) + continue + step.upgrade_messages = step.module.check_and_update_state() if step.upgrade_messages: has_upgrade_messages = True # Any connected input needs to have value DummyDataset (these # are not persisted so we need to do it every time) - module.add_dummy_datasets( connections=step.input_connections ) - # Get the tool - tool = module.tool - # Get the state - step.state = state = module.state - # Get old errors - old_errors = state.inputs.pop( "__errors__", {} ) - # Update the state - step_errors = tool.update_state( trans, tool.inputs, step.state.inputs, step_args, - update_only=True, old_errors=old_errors ) + step.module.add_dummy_datasets( connections=step.input_connections ) + # Store state with the step + step.state = step.module.state + # Error dict + if step.tool_errors: + has_errors = True + errors[step.id] = step.tool_errors else: - # Fix this for multiple inputs - module = step.module = module_factory.from_workflow_step( trans, step ) - state = step.state = module.decode_runtime_state( trans, step_args.pop( "tool_state" ) ) - step_errors = module.update_runtime_state( trans, state, step_args ) - if step_errors: - errors[step.id] = state.inputs["__errors__"] = step_errors - if 'run_workflow' in kwargs and not errors: - new_history = None - if 'new_history' in kwargs: - if 'new_history_name' in kwargs and kwargs['new_history_name'] != '': - nh_name = kwargs['new_history_name'] - else: - nh_name = "History from %s workflow" % workflow.name - if multiple_input_key: - nh_name = '%s %d' % (nh_name, input_number + 1) - new_history = trans.app.model.History( user=trans.user, name=nh_name ) - trans.sa_session.add( new_history ) - # Run each step, connecting outputs to inputs - workflow_invocation = model.WorkflowInvocation() - workflow_invocation.workflow = workflow - outputs = odict() - for i, step in enumerate( workflow.steps ): - # Execute module - job = None - if step.type == 'tool' or step.type is None: - tool = trans.app.toolbox.tools_by_id[ step.tool_id ] - input_values = step.state.inputs - # Connect up - def callback( input, value, prefixed_name, prefixed_label ): - if isinstance( input, DataToolParameter ): - if prefixed_name in step.input_connections_by_name: - conn = step.input_connections_by_name[ prefixed_name ] - return outputs[ conn.output_step.id ][ conn.output_name ] - visit_input_values( tool.inputs, step.state.inputs, callback ) - # Execute it - job, out_data = tool.execute( trans, step.state.inputs, history=new_history) - outputs[ step.id ] = out_data - # Create new PJA associations with the created job, to be run on completion. - # PJA Parameter Replacement (only applies to immediate actions-- rename specifically, for now) - # Pass along replacement dict with the execution of the PJA so we don't have to modify the object. - replacement_dict = {} - for k, v in kwargs.iteritems(): - if k.startswith('wf_parm|'): - replacement_dict[k[8:]] = v - for pja in step.post_job_actions: - if pja.action_type in ActionBox.immediate_actions: - ActionBox.execute(trans.app, trans.sa_session, pja, job, replacement_dict) - else: - job.add_post_job_action(pja) - else: - job, out_data = step.module.execute( trans, step.state ) - outputs[ step.id ] = out_data - # Record invocation - workflow_invocation_step = model.WorkflowInvocationStep() - workflow_invocation_step.workflow_invocation = workflow_invocation - workflow_invocation_step.workflow_step = step - workflow_invocation_step.job = job - # All jobs ran sucessfully, so we can save now - trans.sa_session.add( workflow_invocation ) - invocations.append({'outputs': outputs, - 'new_history': new_history}) - trans.sa_session.flush() - return trans.fill_template( "workflow/run_complete.mako", - workflow=stored, - invocations=invocations ) - else: - # Prepare each step - missing_tools = [] - for step in workflow.steps: - step.upgrade_messages = {} - # Contruct modules - if step.type == 'tool' or step.type is None: - # Restore the tool state for the step - step.module = module_factory.from_workflow_step( trans, step ) - if not step.module: - if step.tool_id not in missing_tools: - missing_tools.append(step.tool_id) - continue - step.upgrade_messages = step.module.check_and_update_state() - if step.upgrade_messages: - has_upgrade_messages = True - # Any connected input needs to have value DummyDataset (these - # are not persisted so we need to do it every time) - step.module.add_dummy_datasets( connections=step.input_connections ) - # Store state with the step - step.state = step.module.state - # Error dict - if step.tool_errors: - has_errors = True - errors[step.id] = step.tool_errors - else: - ## Non-tool specific stuff? - step.module = module_factory.from_workflow_step( trans, step ) - step.state = step.module.get_runtime_state() - # Connections by input name - step.input_connections_by_name = dict( ( conn.input_name, conn ) for conn in step.input_connections ) - if missing_tools: - stored.annotation = self.get_item_annotation_str( trans.sa_session, trans.user, stored ) - return trans.fill_template("workflow/run.mako", steps=[], workflow=stored, missing_tools = missing_tools) - # Render the form - stored.annotation = self.get_item_annotation_str( trans.sa_session, trans.user, stored ) - return trans.fill_template( - "workflow/run.mako", - steps=workflow.steps, - workflow=stored, - has_upgrade_messages=has_upgrade_messages, - errors=errors, - incoming=kwargs ) + ## Non-tool specific stuff? + step.module = module_factory.from_workflow_step( trans, step ) + step.state = step.module.get_runtime_state() + # Connections by input name + step.input_connections_by_name = dict( ( conn.input_name, conn ) for conn in step.input_connections ) + if missing_tools: + stored.annotation = self.get_item_annotation_str( trans.sa_session, trans.user, stored ) + return trans.fill_template("workflow/run.mako", steps=[], workflow=stored, missing_tools = missing_tools) + # Render the form + stored.annotation = self.get_item_annotation_str( trans.sa_session, trans.user, stored ) + return trans.fill_template( + "workflow/run.mako", + steps=workflow.steps, + workflow=stored, + has_upgrade_messages=has_upgrade_messages, + errors=errors, + incoming=kwargs, + history_id=history_id, + hide_fixed_params=hide_fixed_params, + enable_unique_defaults=trans.app.config.enable_unique_workflow_defaults) + finally: + # restore the active history + if saved_history is not None: + trans.set_history(saved_history) def get_item( self, trans, id ): return self.get_stored_workflow( trans, id ) diff --git a/templates/workflow/run.mako b/templates/workflow/run.mako index 7d6dc5c1d4a..afe3f464122 100644 --- a/templates/workflow/run.mako +++ b/templates/workflow/run.mako @@ -6,8 +6,12 @@ %endif - %for i, step in enumerate( steps ): %if step.type == 'tool' or step.type is None: <% tool = app.toolbox.tools_by_id[step.tool_id] %> @@ -355,36 +373,36 @@ if wf_parms: % endif
- ${do_inputs( tool.inputs, step.state.inputs, errors.get( step.id, dict() ), "", step )} - % if step.post_job_actions: -
-
- % if len(step.post_job_actions) > 1: - - % else: - + ${do_inputs( tool.inputs, step.state.inputs, errors.get( step.id, dict() ), "", step, None, used_accumulator )} + % if step.post_job_actions: +
+
+ % if len(step.post_job_actions) > 1: + + % else: + + % endif + <% + pja_ss_all = [] + for pja_ss in [ActionBox.get_short_str(pja) for pja in step.post_job_actions]: + for rematch in re.findall('\$\{.+?\}', pja_ss): + pja_ss = pja_ss.replace(rematch, '%s' % (wf_parms[rematch[2:-1]], rematch[2:-1], rematch[2:-1])) + pja_ss_all.append(pja_ss) + %> + ${'
'.join(pja_ss_all)} +
% endif - <% - pja_ss_all = [] - for pja_ss in [ActionBox.get_short_str(pja) for pja in step.post_job_actions]: - for rematch in re.findall('\$\{.+?\}', pja_ss): - pja_ss = pja_ss.replace(rematch, '%s' % (wf_parms[rematch[2:-1]], rematch[2:-1], rematch[2:-1])) - pja_ss_all.append(pja_ss) - %> - ${'
'.join(pja_ss_all)} -
- % endif +
- - %else: - <% module = step.module %> - -
-
- Step ${int(step.order_index)+1}: ${module.name} - % if step.annotations: -
${step.annotations[0].annotation}
- % endif + %else: + <% module = step.module %> + +
+
+ Step ${int(step.order_index)+1}: ${module.name} + % if step.annotations: +
${step.annotations[0].annotation}
+ % endif
<% @@ -397,7 +415,7 @@ if wf_parms: if not type_filter: type_filter = ['data'] %> - ${do_inputs( module.get_runtime_inputs(type_filter), step.state.inputs, errors.get( step.id, dict() ), "", step )} + ${do_inputs( module.get_runtime_inputs(type_filter), step.state.inputs, errors.get( step.id, dict() ), "", step, None, used_accumulator )}
%endif @@ -411,10 +429,12 @@ if wf_parms: %endfor %else: + %if history_id is None:

named:

+ %endif %endif diff --git a/universe_wsgi.ini.sample b/universe_wsgi.ini.sample index a33f65ad01e..8dbd1002669 100644 --- a/universe_wsgi.ini.sample +++ b/universe_wsgi.ini.sample @@ -444,6 +444,14 @@ use_interactive = True # large servers. #enable_tool_tags = False +# Enable a feature when running workflows. When enabled, default datasets +# are selected for "Set at Runtime" inputs from the history such that the +# same input will not be selected twice, unless there are more inputs than +# compatible datasets in the history. +# When False, the most recently added compatible item in the history will +# be used for each "Set at Runtime" input, independent of others in the Workflow +#enable_unique_workflow_defaults = False + # Enable Galaxy's "Upload via FTP" interface. You'll need to install and # configure an FTP server (we've used ProFTPd since it can use Galaxy's # database for authentication) and set the following two options. @@ -459,6 +467,15 @@ use_interactive = True # Enable enforcement of quotas. Quotas can be set from the Admin interface. #enable_quotas = False +# Enable a feature when running workflows. When enabled, default datasets +# are selected for "Set at Runtime" inputs from the history such that the +# same input will not be selected twice, unless there are more inputs than +# compatible datasets in the history. +# When False, the most recently added compatible item in the history will +# be used for each "Set at Runtime" input, independent of others in the Workflow +#enable_unique_workflow_defaults = False + + # -- Job Execution # If running multiple Galaxy processes, one can be designated as the job From 2eeb2a14aad097d86f84106537c90f118e3641bb Mon Sep 17 00:00:00 2001 From: John Duddy Date: Thu, 18 Aug 2011 14:38:26 -0700 Subject: [PATCH 011/362] Removed duplicate section (merge error) --- universe_wsgi.ini.sample | 8 -------- 1 file changed, 8 deletions(-) diff --git a/universe_wsgi.ini.sample b/universe_wsgi.ini.sample index 8dbd1002669..feee5ad718b 100644 --- a/universe_wsgi.ini.sample +++ b/universe_wsgi.ini.sample @@ -444,14 +444,6 @@ use_interactive = True # large servers. #enable_tool_tags = False -# Enable a feature when running workflows. When enabled, default datasets -# are selected for "Set at Runtime" inputs from the history such that the -# same input will not be selected twice, unless there are more inputs than -# compatible datasets in the history. -# When False, the most recently added compatible item in the history will -# be used for each "Set at Runtime" input, independent of others in the Workflow -#enable_unique_workflow_defaults = False - # Enable Galaxy's "Upload via FTP" interface. You'll need to install and # configure an FTP server (we've used ProFTPd since it can use Galaxy's # database for authentication) and set the following two options. From 4770146fad06cb17108cf465a946f1394cfac24f Mon Sep 17 00:00:00 2001 From: Jeremy Goecks Date: Mon, 3 Oct 2011 17:12:41 -0400 Subject: [PATCH 012/362] Fix typos. --- tools/ngs_rna/cufflinks_wrapper.xml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tools/ngs_rna/cufflinks_wrapper.xml b/tools/ngs_rna/cufflinks_wrapper.xml index 9b727b96264..59c4f122ee6 100644 --- a/tools/ngs_rna/cufflinks_wrapper.xml +++ b/tools/ngs_rna/cufflinks_wrapper.xml @@ -61,10 +61,10 @@ - + - + From f8261fe8063e7be8844fdd067dca315e33b5e596 Mon Sep 17 00:00:00 2001 From: Ross Lazarus Date: Tue, 4 Oct 2011 09:38:33 +1100 Subject: [PATCH 013/362] Increase size limit for display of html files to 10000000 bytes - mirdeep2 creates very large html objects. Size limit for display should probably be configurable. --- lib/galaxy/web/controllers/dataset.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/lib/galaxy/web/controllers/dataset.py b/lib/galaxy/web/controllers/dataset.py index 700823c1f66..6cb7598d369 100644 --- a/lib/galaxy/web/controllers/dataset.py +++ b/lib/galaxy/web/controllers/dataset.py @@ -376,6 +376,8 @@ class DatasetInterface( BaseUIController, UsesAnnotations, UsesHistory, UsesHist raise paste.httpexceptions.HTTPNotFound( "File Not Found (%s)." % data.file_name ) max_peek_size = 1000000 # 1 MB + if isinstance(data.datatype, datatypes.images.Html): + max_peek_size = 10000000 # 10 MB for html if not preview or isinstance(data.datatype, datatypes.images.Image) or os.stat( data.file_name ).st_size < max_peek_size: return open( data.file_name ) else: From a12ef5b45d89117691973d6a9877fd3a67b89687 Mon Sep 17 00:00:00 2001 From: Jeremy Goecks Date: Tue, 4 Oct 2011 10:14:25 -0400 Subject: [PATCH 014/362] Fix bug in setting shared visualization's viewport. --- templates/visualization/display.mako | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/templates/visualization/display.mako b/templates/visualization/display.mako index 2cf9992a1dd..a34b132f71a 100644 --- a/templates/visualization/display.mako +++ b/templates/visualization/display.mako @@ -92,7 +92,8 @@ var callback = function() { view.change_chrom( '${config['viewport']['chrom']}', ${config['viewport']['start']}, ${config['viewport']['end']} ); } %endif view = create_visualization( container_element, "${config.get('title') | h}", - "${config.get('vis_id')}", "${config.get('dbkey')}", callback, + "${config.get('vis_id')}", "${config.get('dbkey')}", + JSON.parse('${ h.to_json_string( config.get( 'viewport', dict() ) ) }'), JSON.parse('${ h.to_json_string( config.get('tracks') ) }'), JSON.parse('${ h.to_json_string( config.get('bookmarks') ) }') ); From 362026e64e4237c655d3444a79ef39c8c4c4af58 Mon Sep 17 00:00:00 2001 From: Jeremy Goecks Date: Tue, 4 Oct 2011 15:56:21 -0400 Subject: [PATCH 015/362] Trackster: add feature-position mapper so that a feature's data can be recovered using from screen position. This is the foundation for providing feature data on mouseover and/or click. --- static/scripts/packed/trackster.js | 2 +- static/scripts/trackster.js | 92 ++++++++++++++++++++++++++---- 2 files changed, 82 insertions(+), 12 deletions(-) diff --git a/static/scripts/packed/trackster.js b/static/scripts/packed/trackster.js index a6885227861..0b6e08e398d 100644 --- a/static/scripts/packed/trackster.js +++ b/static/scripts/packed/trackster.js @@ -1 +1 @@ -var class_module=function(b,a){var c=function(){var f=arguments[0];for(var e=1;ec){a=AFTER}else{if(f<=c){a=CONTAINED_BY}else{a=OVERLAP_END}}}return a};var is_overlap=function(c,b){var a=compute_overlap(c,b);return(a!==BEFORE&&a!==AFTER)};var trackster_module=function(f,aa){var q=f("class").extend,t=f("slotting"),N=f("painters");var ag=function(ah,ai){this.document=ah;this.default_font=ai!==undefined?ai:"9px Monaco, Lucida Console, monospace";this.dummy_canvas=this.new_canvas();this.dummy_context=this.dummy_canvas.getContext("2d");this.dummy_context.font=this.default_font;this.char_width_px=this.dummy_context.measureText("A").width;this.patterns={};this.load_pattern("right_strand","/visualization/strand_right.png");this.load_pattern("left_strand","/visualization/strand_left.png");this.load_pattern("right_strand_inv","/visualization/strand_right_inv.png");this.load_pattern("left_strand_inv","/visualization/strand_left_inv.png")};q(ag.prototype,{load_pattern:function(ah,al){var ai=this.patterns,aj=this.dummy_context,ak=new Image();ak.src=image_path+al;ak.onload=function(){ai[ah]=aj.createPattern(ak,"repeat")}},get_pattern:function(ah){return this.patterns[ah]},new_canvas:function(){var ah=this.document.createElement("canvas");if(window.G_vmlCanvasManager){G_vmlCanvasManager.initElement(ah)}ah.manager=this;return ah}});var o={};var m=function(ah,ai){o[ah.attr("id")]=ai};var n=function(ah,aj,al,ak){al=".group";var ai={};o[ah.attr("id")]=ak;ah.bind("drag",{handle:"."+aj,relative:true},function(au,av){var at=$(this);var ay=$(this).parent(),ap=ay.children(),ar=o[$(this).attr("id")],ao,an,aw,am,aq;an=$(this).parents(al);if(an.length!==0){aw=an.position().top;am=aw+an.outerHeight();if(av.offsetYam){$(this).insertAfter(an);var ax=o[an.attr("id")];ax.remove_drawable(ar);ax.container.add_drawable(ar);return}}}an=null;for(aq=0;aq=aw&&av.offsetY<=am){if(av.offsetY-aw0?aq-1:aq))}}}).bind("dragstart",function(){ai["border-top"]=ah.css("border-top");ai["border-bottom"]=ah.css("border-bottom");$(this).css({"border-top":"1px solid blue","border-bottom":"1px solid blue"})}).bind("dragend",function(){$(this).css(ai)})};aa.moveable=n;var af=16,I=9,F=20,V=I+2,B=100,K=12000,T=200,E=5,w=10,M=5000,x=100,p="There was an error in indexing this dataset. ",L="A converter for this dataset is not installed. Please check your datatypes_conf.xml file.",G="No data for this chrom/contig.",u="Currently indexing... please wait",z="Tool cannot be rerun: ",a="Loading data...",ab="Ready for display",d=10,v=5,D=5;function y(ah){return Math.round(ah*1000)/1000}var c=function(ah){this.num_elements=ah;this.clear()};q(c.prototype,{get:function(ai){var ah=this.key_ary.indexOf(ai);if(ah!==-1){if(this.obj_cache[ai].stale){this.key_ary.splice(ah,1);delete this.obj_cache[ai]}else{this.move_key_to_end(ai,ah)}}return this.obj_cache[ai]},set:function(ai,aj){if(!this.obj_cache[ai]){if(this.key_ary.length>=this.num_elements){var ah=this.key_ary.shift();delete this.obj_cache[ah]}this.key_ary.push(ai)}this.obj_cache[ai]=aj;return aj},move_key_to_end:function(ai,ah){this.key_ary.splice(ah,1);this.key_ary.push(ai)},clear:function(){this.obj_cache={};this.key_ary=[]},size:function(){return this.key_ary.length}});var U=function(ai,ah,aj){c.call(this,ai);this.track=ah;this.subset=(aj!==undefined?aj:true)};q(U.prototype,c.prototype,{load_data:function(aq,al,ao,ai,an){var ap=this.track.view.chrom,ak={chrom:ap,low:aq,high:al,mode:ao,resolution:ai,dataset_id:this.track.dataset_id,hda_ldda:this.track.hda_ldda};$.extend(ak,an);if(this.track.filters_manager){var ar=[];var ah=this.track.filters_manager.filters;for(var am=0;am1){return}return U.prototype.load_data.call(this,aj,ah,al,am,ai,ak)}});var r=function(ak,ai,ah,aj,al){this.name=ak;this.view=ai;this.container=ah;this.drag_handle_class=al;this.config=new H({track:this,params:[{key:"name",label:"Name",type:"text",default_value:ak}],saved_values:aj,onchange:function(){this.track.set_name(this.track.config.values.name)}});this.prefs=this.config.values};q(r.prototype,{init:function(){},request_draw:function(){},_draw:function(){},to_json:function(){},make_name_popup_menu:function(){},set_name:function(ah){this.old_name=this.name;this.name=ah;this.name_div.text(this.name)},revert_name:function(){this.name=this.old_name;this.name_div.text(this.name)},remove:function(){this.container.remove_drawable(this);this.container_div.fadeOut("slow",function(){$(this).remove();view.update_intro_div();view.has_changes=true})}});var A=function(al,ak,ai,ah,aj,am){r.call(this,ak,ai,ah,aj,am);this.obj_type=al;this.drawables=[]};q(A.prototype,r.prototype,{init:function(){for(var ah=0;ah").addClass("group").attr("id","group_"+al).appendTo(this.container.content_div);this.header_div=$("
").addClass("track-header").appendTo(this.container_div);this.header_div.append($("
").addClass(this.drag_handle_class));this.name_div=$("
").addClass("group-name menubutton popup").text(this.name).appendTo(this.header_div);this.content_div=$("
").addClass("content-div").attr("id","group_"+al+"_content_div").appendTo(this.container_div);m(this.container_div,this);m(this.content_div,this);n(this.container_div,this.drag_handle_class,".group",this);this.make_name_popup_menu()};q(S.prototype,r.prototype,A.prototype,{make_name_popup_menu:function(){var ai=this;var ah={};ah["Edit configuration"]=function(){var al=function(){hide_modal();$(window).unbind("keypress.check_enter_esc")},aj=function(){ai.config.update_from_form($(".dialog-box"));hide_modal();$(window).unbind("keypress.check_enter_esc")},ak=function(am){if((am.keyCode||am.which)===27){al()}else{if((am.keyCode||am.which)===13){aj()}}};$(window).bind("keypress.check_enter_esc",ak);show_modal("Configure Group",ai.config.build_form(),{Cancel:al,OK:aj})};ah.Remove=function(){ai.remove()};make_popupmenu(ai.name_div,ah)}});var ae=function(ah,ak,aj,ai){A.call(this,"View");this.container=ah;this.chrom=null;this.vis_id=aj;this.dbkey=ai;this.title=ak;this.tracks=this.drawables;this.label_tracks=[];this.tracks_to_be_redrawn=[];this.max_low=0;this.max_high=0;this.zoom_factor=3;this.min_separation=30;this.has_changes=false;this.load_chroms_deferred=null;this.init();this.canvas_manager=new ag(ah.get(0).ownerDocument);this.reset()};q(ae.prototype,A.prototype,{init:function(){var aj=this.container,ah=this;this.top_container=$("
").addClass("top-container").appendTo(aj);this.browser_content_div=$("
").addClass("content").css("position","relative").appendTo(aj);this.bottom_container=$("
").addClass("bottom-container").appendTo(aj);this.top_labeltrack=$("
").addClass("top-labeltrack").appendTo(this.top_container);this.viewport_container=$("
").addClass("viewport-container").attr("id","viewport-container").appendTo(this.browser_content_div);this.content_div=this.viewport_container;m(this.viewport_container,ah);this.intro_div=$("
").addClass("intro");var ak=$("
").text("Add Datasets to Visualization").addClass("action-button").appendTo(this.intro_div).click(function(){add_tracks()});this.nav_labeltrack=$("
").addClass("nav-labeltrack").appendTo(this.bottom_container);this.nav_container=$("
").addClass("nav-container").prependTo(this.top_container);this.nav=$("
").addClass("nav").appendTo(this.nav_container);this.overview=$("
").addClass("overview").appendTo(this.bottom_container);this.overview_viewport=$("
").addClass("overview-viewport").appendTo(this.overview);this.overview_close=$("Close Overview").addClass("overview-close").hide().appendTo(this.overview_viewport);this.overview_highlight=$("
").addClass("overview-highlight").hide().appendTo(this.overview_viewport);this.overview_box_background=$("
").addClass("overview-boxback").appendTo(this.overview_viewport);this.overview_box=$("
").addClass("overview-box").appendTo(this.overview_viewport);this.default_overview_height=this.overview_box.height();this.nav_controls=$("
").addClass("nav-controls").appendTo(this.nav);this.chrom_select=$("").addClass("nav-input").hide().bind("keyup focusout",ai).appendTo(this.nav_controls);this.location_span=$("").addClass("location").appendTo(this.nav_controls);this.location_span.click(function(){ah.location_span.hide();ah.chrom_select.hide();ah.nav_input.val(ah.chrom+":"+ah.low+"-"+ah.high);ah.nav_input.css("display","inline-block");ah.nav_input.select();ah.nav_input.focus()});if(this.vis_id!==undefined){this.hidden_input=$("").attr("type","hidden").val(this.vis_id).appendTo(this.nav_controls)}this.zo_link=$("").click(function(){ah.zoom_out();ah.request_redraw()}).appendTo(this.nav_controls);this.zi_link=$("").click(function(){ah.zoom_in();ah.request_redraw()}).appendTo(this.nav_controls);this.load_chroms_deferred=this.load_chroms({low:0});this.chrom_select.bind("change",function(){ah.change_chrom(ah.chrom_select.val())});this.browser_content_div.click(function(al){$(this).find("input").trigger("blur")});this.browser_content_div.bind("dblclick",function(al){ah.zoom_in(al.pageX,this.viewport_container)});this.overview_box.bind("dragstart",function(al,am){this.current_x=am.offsetX}).bind("drag",function(al,an){var ao=an.offsetX-this.current_x;this.current_x=an.offsetX;var am=Math.round(ao/ah.viewport_container.width()*(ah.max_high-ah.max_low));ah.move_delta(-am)});this.overview_close.click(function(){ah.reset_overview()});this.viewport_container.bind("draginit",function(al,am){if(al.clientX>ah.viewport_container.width()-16){return false}}).bind("dragstart",function(al,am){am.original_low=ah.low;am.current_height=al.clientY;am.current_x=am.offsetX}).bind("drag",function(an,ap){var al=$(this);var aq=ap.offsetX-ap.current_x;var am=al.scrollTop()-(an.clientY-ap.current_height);al.scrollTop(am);ap.current_height=an.clientY;ap.current_x=ap.offsetX;var ao=Math.round(aq/ah.viewport_container.width()*(ah.high-ah.low));ah.move_delta(ao)}).bind("mousewheel",function(an,ap,am,al){if(am){var ao=Math.round(-am/ah.viewport_container.width()*(ah.high-ah.low));ah.move_delta(ao)}});this.top_labeltrack.bind("dragstart",function(al,am){return $("
").css({height:ah.browser_content_div.height()+ah.top_labeltrack.height()+ah.nav_labeltrack.height()+1,top:"0px",position:"absolute","background-color":"#ccf",opacity:0.5,"z-index":1000}).appendTo($(this))}).bind("drag",function(ap,aq){$(aq.proxy).css({left:Math.min(ap.pageX,aq.startX),width:Math.abs(ap.pageX-aq.startX)});var am=Math.min(ap.pageX,aq.startX)-ah.container.offset().left,al=Math.max(ap.pageX,aq.startX)-ah.container.offset().left,ao=(ah.high-ah.low),an=ah.viewport_container.width();ah.update_location(Math.round(am/an*ao)+ah.low,Math.round(al/an*ao)+ah.low)}).bind("dragend",function(aq,ar){var am=Math.min(aq.pageX,ar.startX),al=Math.max(aq.pageX,ar.startX),ao=(ah.high-ah.low),an=ah.viewport_container.width(),ap=ah.low;ah.low=Math.round(am/an*ao)+ap;ah.high=Math.round(al/an*ao)+ap;$(ar.proxy).remove();ah.request_redraw()});this.add_label_track(new ad(this,{content_div:this.top_labeltrack}));this.add_label_track(new ad(this,{content_div:this.nav_labeltrack}));$(window).bind("resize",function(){ah.resize_window()});$(document).bind("redraw",function(){ah.redraw()});this.reset();$(window).trigger("resize");this.update_intro_div()},update_intro_div:function(){if(this.num_tracks===0){this.intro_div.appendTo(this.viewport_container)}else{this.intro_div.remove()}},update_location:function(ah,ai){this.location_span.text(commatize(ah)+" - "+commatize(ai));this.nav_input.val(this.chrom+":"+commatize(ah)+"-"+commatize(ai))},load_chroms:function(aj){aj.num=x;$.extend(aj,(this.vis_id!==undefined?{vis_id:this.vis_id}:{dbkey:this.dbkey}));var ah=this,ai=$.Deferred();$.ajax({url:chrom_url,data:aj,dataType:"json",success:function(al){if(al.chrom_info.length===0){alert("Invalid chromosome: "+aj.chrom);return}if(al.reference){ah.add_label_track(new C(ah))}ah.chrom_data=al.chrom_info;var ao='';for(var an=0,ak=ah.chrom_data.length;an'+am+""}if(al.prev_chroms){ao+='"}if(al.next_chroms){ao+='"}ah.chrom_select.html(ao);ah.chrom_start_index=al.start_index;ai.resolve(al)},error:function(){alert("Could not load chroms for this dbkey:",ah.dbkey)}});return ai},change_chrom:function(al,ai,an){if(!al||al==="None"){return}var ak=this;if(al==="previous"){ak.load_chroms({low:this.chrom_start_index-x});return}if(al==="next"){ak.load_chroms({low:this.chrom_start_index+x});return}var am=$.grep(ak.chrom_data,function(ap,aq){return ap.chrom===al})[0];if(am===undefined){ak.load_chroms({chrom:al},function(){ak.change_chrom(al,ai,an)});return}else{if(al!==ak.chrom){ak.chrom=al;ak.chrom_select.val(ak.chrom);ak.max_high=am.len-1;ak.reset();ak.request_redraw(true);for(var ao=0,ah=ak.tracks.length;aoah.max_high){ah.high=ah.max_high;ah.low=ah.max_high-ai}else{ah.high-=aj;ah.low-=aj}}ah.request_redraw()},add_drawable:function(ah){A.prototype.add_drawable.call(this,ah);if(ah.init){ah.init()}this.has_changes=true;this.update_intro_div()},add_label_track:function(ah){ah.view=this;this.label_tracks.push(ah)},remove_drawable:function(aj,ai){A.prototype.remove_drawable.call(this,aj);if(ai){var ah=this;aj.container_div.fadeOut("slow",function(){$(this).remove();ah.update_intro_div()});this.has_changes=true}},reset:function(){this.low=this.max_low;this.high=this.max_high;this.viewport_container.find(".yaxislabel").remove()},request_redraw:function(ap,ah,ao,ai){var an=this,al=(ai?[ai]:an.tracks),aj;var ai;for(var am=0;amthis.max_high){ak=this.max_high}var aq=this.high-this.low;if(this.high!==0&&aq").addClass("dynamic-tool").hide();this.parent_div.bind("drag",function(ay){ay.stopPropagation()}).click(function(ay){ay.stopPropagation()}).bind("dblclick",function(ay){ay.stopPropagation()});var at=$("
").appendTo(this.parent_div).text(this.name);var aq=this.params;var ao=this;$.each(this.params,function(az,aC){var aB=$("
").addClass("param-row").appendTo(ao.parent_div);var ay=$("
").addClass("param-label").text(aC.label).appendTo(aB);var aA=$("
").addClass("slider").html(aC.html).appendTo(aB);aA.find(":input").val(aC.value);$("
").appendTo(aB)});this.parent_div.find("input").click(function(){$(this).select()});var ax=$("
").addClass("param-row").appendTo(this.parent_div);var am=$("").attr("value","Run on complete dataset").appendTo(ax);var ah=$("").attr("value","Run on visible region").css("margin-left","3em").appendTo(ax);var ao=this;ah.click(function(){ao.run_on_region()});am.click(function(){ao.run_on_dataset()})};q(s.prototype,{get_param_values_dict:function(){var ah={};this.parent_div.find(":input").each(function(){var ai=$(this).attr("name"),aj=$(this).val();ah[ai]=JSON.stringify(aj)});return ah},get_param_values:function(){var ai=[];var ah={};this.parent_div.find(":input").each(function(){var aj=$(this).attr("name"),ak=$(this).val();if(aj){ai[ai.length]=ak}});return ai},run_on_dataset:function(){var ah=this;ah.run({dataset_id:this.track.original_dataset_id,tool_id:ah.name},null,function(ai){show_modal(ah.name+" is Running",ah.name+" is running on the complete dataset. Tool outputs are in dataset's history.",{Close:hide_modal})})},run_on_region:function(){var ai={dataset_id:this.track.original_dataset_id,chrom:this.track.view.chrom,low:this.track.view.low,high:this.track.view.high,tool_id:this.name},al=this.track,aj=ai.tool_id+al.tool_region_and_parameters_str(ai.chrom,ai.low,ai.high),ah,am;if(al.container===view){var ak=new S(this.name,this.track.view,this.track.container);al.container.add_drawable(ak);al.container.remove_drawable(al);ak.add_drawable(al);al.container_div.appendTo(ak.content_div);ah=ak}else{ah=al.container}if(al instanceof e){am=new X(aj,view,ah,"hda");am.change_mode(al.mode);ah.add_drawable(am)}am.content_div.text("Starting job.");this.run(ai,am,function(an){am.dataset_id=an.dataset_id;am.content_div.text("Running job.");am.init()})},run:function(ai,aj,ak){$.extend(ai,this.get_param_values_dict());var ah=function(){$.getJSON(rerun_tool_url,ai,function(al){if(al==="no converter"){aj.container_div.addClass("error");aj.content_div.text(L)}else{if(al.error){aj.container_div.addClass("error");aj.content_div.text(z+al.message)}else{if(al==="pending"){aj.container_div.addClass("pending");aj.content_div.text("Converting input data so that it can be used quickly with tool.");setTimeout(ah,2000)}else{ak(al)}}}})};ah()}});var P=function(ai,ah,aj,ak){this.name=ai;this.label=ah;this.html=aj;this.value=ak};var g=function(aj,ai,al,am,ak,ah){P.call(this,aj,ai,al,am);this.min=ak;this.max=ah};var h=function(ai,ah,aj,ak){this.name=ai;this.index=ah;this.tool_id=aj;this.tool_exp_name=ak};var Y=function(ai,ah,aj,ak){h.call(this,ai,ah,aj,ak);this.low=-Number.MAX_VALUE;this.high=Number.MAX_VALUE;this.min=Number.MAX_VALUE;this.max=-Number.MAX_VALUE;this.container=null;this.slider=null;this.slider_label=null};q(Y.prototype,{applies_to:function(ah){if(ah.length>this.index){return true}return false},keep:function(ah){if(!this.applies_to(ah)){return true}var ai=parseFloat(ah[this.index]);return(isNaN(ai)||(ai>=this.low&&ai<=this.high))},update_attrs:function(ai){var ah=false;if(!this.applies_to(ai)){return ah}if(ai[this.index]this.max){this.max=Math.ceil(ai[this.index]);ah=true}return ah},update_ui_elt:function(){if(this.min!=this.max){this.container.show()}else{this.container.hide()}var aj=function(am,ak){var al=ak-am;return(al<=2?0.01:1)};var ai=this.slider.slider("option","min"),ah=this.slider.slider("option","max");if(this.minah){this.slider.slider("option","min",this.min);this.slider.slider("option","max",this.max);this.slider.slider("option","step",aj(this.min,this.max));this.slider.slider("option","values",[this.min,this.max])}}});var ac=function(ar,ay){this.track=ar;this.filters=[];for(var at=0;at").attr("size",input_size).attr("maxlength",input_size).attr("value",aD).appendTo(aB).focus().select().click(function(aE){aE.stopPropagation()}).blur(function(){$(this).remove();aB.text(aD)}).keyup(function(aI){if(aI.keyCode===27){$(this).trigger("blur")}else{if(aI.keyCode===13){var aG=aC.slider("option","min"),aE=aC.slider("option","max"),aH=function(aJ){return(isNaN(aJ)||aJ>aE||aJ").addClass("filters").hide();this.parent_div.bind("drag",function(aA){aA.stopPropagation()}).click(function(aA){aA.stopPropagation()}).bind("dblclick",function(aA){aA.stopPropagation()}).bind("keydown",function(aA){aA.stopPropagation()});var av=$("
").addClass("sliders").appendTo(this.parent_div);var ap=this;$.each(this.filters,function(aD,aF){aF.container=$("
").addClass("slider-row").appendTo(av);var aE=$("
").addClass("elt-label").appendTo(aF.container);var aC=$("").addClass("slider-name").text(aF.name+" ").appendTo(aE);var aB=$("");var aH=$("").addClass("slider-value").appendTo(aE).append("[").append(aB).append("]");var aA=$("
").addClass("slider").appendTo(aF.container);aF.control_element=$("
").attr("id",aF.name+"-filter-control").appendTo(aA);var aG=[0,0];aF.control_element.slider({range:true,min:Number.MAX_VALUE,max:-Number.MIN_VALUE,values:[0,0],slide:function(aJ,aK){var aI=aK.values;aB.text(aI[0]+"-"+aI[1]);aF.low=aI[0];aF.high=aI[1];ap.track.request_draw(true,true)},change:function(aI,aJ){aF.control_element.slider("option","slide").call(aF.control_element,aI,aJ)}});aF.slider=aF.control_element;aF.slider_label=aB;al(aH,aB,aF.control_element);$("
").appendTo(aF.container)});if(this.filters.length!==0){var am=$("
").addClass("param-row").appendTo(av);var ao=$("").attr("value","Run on complete dataset").appendTo(am);var aj=this;ao.click(function(){aj.run_on_dataset()})}var aq=$("
").addClass("display-controls").appendTo(this.parent_div),an=$("").addClass("elt-label").text("Transparency:").appendTo(aq),ai=$("').attr("id",aj).attr("name",aj).attr("checked",ao))}else{if(ak.type==="text"){ar.append($('').attr("id",aj).val(ao).click(function(){$(this).select()}))}else{if(ak.type==="color"){var an=$("").attr("id",aj).attr("name",aj).val(ao);var ap=$("
").hide();var al=$("
").appendTo(ap);var aq=$("
").appendTo(al).farbtastic({width:100,height:100,callback:an,color:ao});$("
").append(an).append(ap).appendTo(ar).bind("click",function(at){ap.css({left:$(this).position().left+($(an).width()/2)-60,top:$(this).position().top+$(this.height)}).show();$(document).bind("click.color-picker",function(){ap.hide();$(document).unbind("click.color-picker")});at.stopPropagation()})}else{ar.append($("").attr("id",aj).attr("name",aj).val(ao))}}}}});return ah},update_from_form:function(ah){var aj=this;var ai=false;$.each(this.params,function(ak,am){if(!am.hidden){var an="param_"+ak;var al=ah.find("#"+an).val();if(am.type==="float"){al=parseFloat(al)}else{if(am.type==="int"){al=parseInt(al)}else{if(am.type==="bool"){al=ah.find("#"+an).is(":checked")}}}if(al!==aj.values[am.key]){aj.values[am.key]=al;ai=true}}});if(ai){this.onchange()}}});var b=function(aj,ai,ah,ak){this.index=aj;this.low=aj*T*ai;this.high=(aj+1)*T*ai;this.resolution=ai;this.canvas=$("
").append(ah);this.data=ak;this.stale=false};var l=function(aj,ai,ah,ak,al){b.call(this,aj,ai,ah,ak);this.max_val=al};var R=function(aj,ai,ah,al,ak){b.call(this,aj,ai,ah,al);this.message=ak};var j=function(ak,ai,ah,aj,al,am){r.call(this,ak,ai,ah,{},"draghandle");this.data_url=(al?al:default_data_url);this.data_url_extra_params={};this.data_query_wait=(am?am:M);this.dataset_check_url=converted_datasets_state_url;if(!j.id_counter){j.id_counter=0}this.container_div=$("
").addClass("track").attr("id","track_"+j.id_counter++).css("position","relative");if(!this.hidden){this.header_div=$("
").appendTo(this.container_div);if(this.view.editor){this.drag_div=$("
").addClass(this.drag_handle_class).appendTo(this.header_div)}this.name_div=$("