From 4df1de38ba390ee6513bc85fbb0643e6e152bd25 Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Mon, 4 Feb 2019 23:46:25 +0100 Subject: [PATCH 1/3] Map over multiple=true in workflow run There is some unfortunate divergence between the mapping over code in tool runs vs workflow runs. Without this patch a list:list input to a multiple="true" data input would get fully reduced in workflow, while tool runs would just reduce the inner collection (and I think that's the correct behavior). --- lib/galaxy/workflow/modules.py | 26 +++++++++++++++++--------- 1 file changed, 17 insertions(+), 9 deletions(-) diff --git a/lib/galaxy/workflow/modules.py b/lib/galaxy/workflow/modules.py index c7a56f4d841..34c079074b1 100644 --- a/lib/galaxy/workflow/modules.py +++ b/lib/galaxy/workflow/modules.py @@ -314,13 +314,10 @@ class WorkflowModule(object): def _find_collections_to_match(self, progress, step, all_inputs): collections_to_match = matching.CollectionsToMatch() + dataset_collection_type_descriptions = self.trans.app.dataset_collections_service.collection_type_descriptions for input_dict in all_inputs: name = input_dict["name"] - multiple = input_dict["multiple"] - if multiple: - continue - data = progress.replacement_for_input(step, input_dict) can_map_over = hasattr(data, "collection") # and data.collection.allow_implicit_mapping @@ -329,13 +326,23 @@ class WorkflowModule(object): is_data_param = input_dict["input_type"] == "dataset" if is_data_param: - collections_to_match.add(name, data) + multiple = input_dict["multiple"] + if multiple: + # multiple="true" data input, acts like "list" collection_type. + # just need to figure out subcollection_type_description + history_query = HistoryQuery.from_collection_types( + ['list'], + dataset_collection_type_descriptions, + ) + subcollection_type_description = history_query.can_map_over(data) + if subcollection_type_description: + collections_to_match.add(name, data, subcollection_type=subcollection_type_description) + else: + collections_to_match.add(name, data) continue is_data_collection_param = input_dict["input_type"] == "dataset_collection" if is_data_collection_param: - dataset_collection_type_descriptions = self.trans.app.dataset_collections_service.collection_type_descriptions - history_query = HistoryQuery.from_collection_types( input_dict.get("collection_types", None), dataset_collection_type_descriptions, @@ -1158,8 +1165,9 @@ class ToolModule(WorkflowModule): replacement = NO_REPLACEMENT if iteration_elements and prefixed_name in iteration_elements: - if isinstance(input, DataToolParameter): - # Pull out dataset instance from element. + if isinstance(input, DataToolParameter) and hasattr(iteration_elements[prefixed_name], 'dataset_instance'): + # Pull out dataset instance (=HDA) from element and set a temporary element_identifier attribute + # See https://github.com/galaxyproject/galaxy/pull/1693 for context. replacement = iteration_elements[prefixed_name].dataset_instance if hasattr(iteration_elements[prefixed_name], u'element_identifier') and iteration_elements[prefixed_name].element_identifier: replacement.element_identifier = iteration_elements[prefixed_name].element_identifier From cf68cbe907729283a8a359d9a8dbcb0013ebcee5 Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Tue, 5 Feb 2019 12:37:40 +0100 Subject: [PATCH 2/3] Verify that list:list will be reduced to list in multi data input --- test/api/test_workflows.py | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/test/api/test_workflows.py b/test/api/test_workflows.py index b0c2d33e1d2..5b46d5c2da6 100644 --- a/test/api/test_workflows.py +++ b/test/api/test_workflows.py @@ -766,6 +766,33 @@ test_data: replaced_hda = self.dataset_populator.get_history_dataset_details(history_id, dataset_id=replaced_hda_id, wait=True, assert_ok=False) assert not replaced_hda['visible'], replaced_hda + @skip_without_tool('multi_data_optional') + def test_workflow_list_list_multi_data_map_over(self): + # Test that a list:list is reduced to list with a multiple="true" data input + with self.dataset_populator.test_history() as history_id: + workflow_id = self._upload_yaml_workflow(""" +class: GalaxyWorkflow +inputs: + input_datasets: collection +steps: + multi_data_optional: + tool_id: multi_data_optional + in: + input1: input_datasets +""") + with self.dataset_populator.test_history() as history_id: + hdca_id = self.dataset_collection_populator.create_list_of_list_in_history(history_id).json() + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + inputs = { + '0': self._ds_entry(hdca_id), + } + invocation_id = self.__invoke_workflow(history_id, workflow_id, inputs) + self.wait_for_invocation_and_jobs(history_id, workflow_id, invocation_id) + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + output_collection = self.dataset_populator.get_history_collection_details(history_id, hid=6) + assert output_collection['collection_type'] == 'list' + assert output_collection['job_source_type'] == 'ImplicitCollectionJobs' + @skip_without_tool("cat_list") @skip_without_tool("collection_creates_pair") def test_workflow_run_output_collection_mapping(self): From a36b56e6940cf8e406ebca90114e2a9974e50e0d Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Tue, 5 Feb 2019 15:39:35 +0100 Subject: [PATCH 3/3] Drop redundant wait for history wait_for_invocation_and_jobs does that already, thx @nsoranzo. --- test/api/test_workflows.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/test/api/test_workflows.py b/test/api/test_workflows.py index 5b46d5c2da6..7d70f570029 100644 --- a/test/api/test_workflows.py +++ b/test/api/test_workflows.py @@ -788,7 +788,6 @@ steps: } invocation_id = self.__invoke_workflow(history_id, workflow_id, inputs) self.wait_for_invocation_and_jobs(history_id, workflow_id, invocation_id) - self.dataset_populator.wait_for_history(history_id, assert_ok=True) output_collection = self.dataset_populator.get_history_collection_details(history_id, hid=6) assert output_collection['collection_type'] == 'list' assert output_collection['job_source_type'] == 'ImplicitCollectionJobs' @@ -805,7 +804,6 @@ steps: } invocation_id = self.__invoke_workflow(history_id, workflow_id, inputs) self.wait_for_invocation_and_jobs(history_id, workflow_id, invocation_id) - self.dataset_populator.wait_for_history(history_id, assert_ok=True) self.assertEqual("a\nc\nb\nd\ne\ng\nf\nh\n", self.dataset_populator.get_history_dataset_content(history_id, hid=0)) @skip_without_tool("cat_list")