From d3f7510b2db932228c76402062bad9095549f5f0 Mon Sep 17 00:00:00 2001 From: Nicola Soranzo Date: Wed, 9 Nov 2016 17:04:51 +0000 Subject: [PATCH] Remove unnecessary set_output_history parameter Also fix import order and Python3 compatibility. --- .ci/flake8_lint_include_list.txt | 1 + lib/galaxy/datatypes/data.py | 4 +-- lib/galaxy/model/__init__.py | 6 ++-- lib/galaxy/tools/actions/__init__.py | 42 +++++++++++----------------- 4 files changed, 23 insertions(+), 30 deletions(-) diff --git a/.ci/flake8_lint_include_list.txt b/.ci/flake8_lint_include_list.txt index e09886af693..e10870983a9 100644 --- a/.ci/flake8_lint_include_list.txt +++ b/.ci/flake8_lint_include_list.txt @@ -264,6 +264,7 @@ lib/galaxy/sample_tracking/__init__.py lib/galaxy/sample_tracking/sample.py lib/galaxy/security/validate_user_input.py lib/galaxy/tags/ +lib/galaxy/tools/actions/__init__.py lib/galaxy/tools/actions/metadata.py lib/galaxy/tools/cwl/ lib/galaxy/tools/data_manager/__init__.py diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index 0fbcaaaf07a..2c6ff80ce84 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -502,7 +502,7 @@ class Data( object ): """Returns ( target_ext, existing converted dataset )""" return datatypes_registry.find_conversion_destination_for_dataset_by_extensions( dataset, accepted_formats, **kwd ) - def convert_dataset(self, trans, original_dataset, target_type, return_output=False, visible=True, deps=None, set_output_history=True, target_context=None): + def convert_dataset(self, trans, original_dataset, target_type, return_output=False, visible=True, deps=None, target_context=None): """This function adds a job to the queue to convert a dataset to another type. Returns a message about success/failure.""" converter = trans.app.datatypes_registry.get_converter_by_target_type( original_dataset.ext, target_type ) @@ -525,7 +525,7 @@ class Data( object ): params[input_name] = original_dataset # Run converter, job is dispatched through Queue - converted_dataset = converter.execute( trans, incoming=params, set_output_hid=visible, set_output_history=set_output_history)[1] + converted_dataset = converter.execute( trans, incoming=params, set_output_hid=visible )[1] if len(params) > 0: trans.log_event( "Converter params: %s" % (str(params)), tool_id=converter.id ) if not visible: diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 8f8f9cf44c1..4794ce037f6 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -2080,7 +2080,7 @@ class DatasetInstance( object ): raise NoConverterException("A dependency (%s) is missing a converter." % dependency) except KeyError: pass # No deps - new_dataset = next(iter(self.datatype.convert_dataset( trans, self, target_ext, return_output=True, visible=False, deps=deps, set_output_history=True, target_context=target_context ).values())) + new_dataset = next(iter(self.datatype.convert_dataset( trans, self, target_ext, return_output=True, visible=False, deps=deps, target_context=target_context ).values())) new_dataset.name = self.name self.copy_attributes( new_dataset ) assoc = ImplicitlyConvertedDatasetAssociation( parent=self, file_type=target_ext, dataset=new_dataset, metadata_safe=False ) @@ -3139,14 +3139,14 @@ class DatasetCollection( object, Dictifiable, UsesAnnotations ): def populated( self ): top_level_populated = self.populated_state == DatasetCollection.populated_states.OK if top_level_populated and self.has_subcollections: - return all(map(lambda e: e.child_collection.populated, self.elements)) + return all(e.child_collection.populated for e in self.elements) return top_level_populated @property def waiting_for_elements( self ): top_level_waiting = self.populated_state == DatasetCollection.populated_states.NEW if not top_level_waiting and self.has_subcollections: - return any(map(lambda e: e.child_collection.waiting_for_elements, self.elements)) + return any(e.child_collection.waiting_for_elements for e in self.elements) return top_level_waiting def mark_as_populated( self ): diff --git a/lib/galaxy/tools/actions/__init__.py b/lib/galaxy/tools/actions/__init__.py index a21ec9fdfda..e4ff04a46c2 100644 --- a/lib/galaxy/tools/actions/__init__.py +++ b/lib/galaxy/tools/actions/__init__.py @@ -1,4 +1,5 @@ import json +import logging import re from json import dumps @@ -7,16 +8,15 @@ from six import string_types from galaxy import model from galaxy.exceptions import ObjectInvalid from galaxy.model import LibraryDatasetDatasetAssociation +from galaxy.tools.parameters import update_param from galaxy.tools.parameters.basic import DataCollectionToolParameter, DataToolParameter, RuntimeValue from galaxy.tools.parameters.wrapped import WrappedParameters -from galaxy.tools.parameters import update_param from galaxy.util import ExecutionTimer from galaxy.util.none_like import NoneDataset from galaxy.util.odict import odict from galaxy.util.template import fill_template from galaxy.web import url_for -import logging log = logging.getLogger( __name__ ) @@ -193,7 +193,7 @@ class DefaultToolAction( object ): return history, inp_data, inp_dataset_collections - def execute(self, tool, trans, incoming={}, return_job=False, set_output_hid=True, set_output_history=True, history=None, job_params=None, rerun_remap_job_id=None, mapping_over_collection=False, execution_cache=None ): + def execute(self, tool, trans, incoming={}, return_job=False, set_output_hid=True, history=None, job_params=None, rerun_remap_job_id=None, mapping_over_collection=False, execution_cache=None ): """ Executes a tool, creating job and tool outputs, associating them, and submitting the job to the job queue. If history is not specified, use @@ -254,7 +254,7 @@ class DefaultToolAction( object ): wrapped_params = self._wrapped_params( trans, tool, incoming ) out_data = odict() - input_collections = dict( [ (k, v[0][0]) for k, v in inp_dataset_collections.iteritems() ] ) + input_collections = dict( (k, v[0][0]) for k, v in inp_dataset_collections.items() ) output_collections = OutputCollections( trans, history, @@ -345,9 +345,6 @@ class DefaultToolAction( object ): if not filter_output(output, incoming): if output.collection: collections_manager = app.dataset_collections_service - # As far as I can tell - this is always true - but just verify - assert set_output_history, "Cannot create dataset collection for this kind of tool." - element_identifiers = [] known_outputs = output.known_outputs( input_collections, collections_manager.type_registry ) # Just to echo TODO elsewhere - this should be restructured to allow @@ -360,7 +357,7 @@ class DefaultToolAction( object ): for parent_id in (output_part_def.parent_ids or []): # TODO: replace following line with formal abstractions for doing this. current_collection_type = ":".join(current_collection_type.split(":")[1:]) - name_to_index = dict(map(lambda (index, value): (value["name"], index), enumerate(current_element_identifiers))) + name_to_index = dict((value["name"], index) for (index, value) in enumerate(current_element_identifiers)) if parent_id not in name_to_index: if parent_id not in current_element_identifiers: index = len(current_element_identifiers) @@ -382,8 +379,7 @@ class DefaultToolAction( object ): # Following hack causes dataset to no be added to history... child_dataset_names.add( effective_output_name ) - if set_output_history: - history.add_dataset( element, set_hid=set_output_hid, quota=False ) + history.add_dataset( element, set_hid=set_output_hid, quota=False ) trans.sa_session.add( element ) trans.sa_session.flush() @@ -416,15 +412,11 @@ class DefaultToolAction( object ): if name not in child_dataset_names and name not in incoming: # don't add children; or already existing datasets, i.e. async created data = out_data[ name ] datasets_to_persist.append( data ) - if set_output_history: - # Set HID and add to history. - # This is brand new and certainly empty so don't worry about quota. - # TOOL OPTIMIZATION NOTE - from above loop to the job create below 99%+ - # of execution time happens within in history.add_datasets. - history.add_datasets( trans.sa_session, datasets_to_persist, set_hid=set_output_hid, quota=False, flush=False ) - else: - for data in datasets_to_persist: - trans.sa_session.add( data ) + # Set HID and add to history. + # This is brand new and certainly empty so don't worry about quota. + # TOOL OPTIMIZATION NOTE - from above loop to the job create below 99%+ + # of execution time happens within in history.add_datasets. + history.add_datasets( trans.sa_session, datasets_to_persist, set_hid=set_output_hid, quota=False, flush=False ) # Add all the children to their parents for parent_name, child_name in parent_to_child_pairs: @@ -549,7 +541,7 @@ class DefaultToolAction( object ): # FIXME: Don't need all of incoming here, just the defined parameters # from the tool. We need to deal with tools that pass all post # parameters to the command as a special case. - for name, dataset_collection_info_pairs in inp_dataset_collections.iteritems(): + for name, dataset_collection_info_pairs in inp_dataset_collections.items(): first_reduction = True for ( dataset_collection, reduced ) in dataset_collection_info_pairs: # TODO: update incoming for list... @@ -562,23 +554,23 @@ class DefaultToolAction( object ): # datasets below? # TODO: verify can have multiple with same name, don't want to loose tracability job.add_input_dataset_collection( name, dataset_collection ) - for name, value in tool.params_to_strings( incoming, trans.app ).iteritems(): + for name, value in tool.params_to_strings( incoming, trans.app ).items(): job.add_parameter( name, value ) self._check_input_data_access( trans, job, inp_data, current_user_roles ) def _record_outputs( self, job, out_data, output_collections ): out_collections = output_collections.out_collections out_collection_instances = output_collections.out_collection_instances - for name, dataset in out_data.iteritems(): + for name, dataset in out_data.items(): job.add_output_dataset( name, dataset ) - for name, dataset_collection in out_collections.iteritems(): + for name, dataset_collection in out_collections.items(): job.add_implicit_output_dataset_collection( name, dataset_collection ) - for name, dataset_collection_instance in out_collection_instances.iteritems(): + for name, dataset_collection_instance in out_collection_instances.items(): job.add_output_dataset_collection( name, dataset_collection_instance ) def _check_input_data_access( self, trans, job, inp_data, current_user_roles ): access_timer = ExecutionTimer() - for name, dataset in inp_data.iteritems(): + for name, dataset in inp_data.items(): if dataset: if not trans.app.security_agent.can_access_dataset( current_user_roles, dataset.dataset ): raise Exception("User does not have permission to use a dataset (%s) provided for input." % dataset.id)