From 9d6a0e82f6f9634981ea96c37f0f48e38e9c5b12 Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Wed, 2 Aug 2017 09:22:21 +0200 Subject: [PATCH 1/4] Implement default_identifier_source for outputs This allows tool authors to specify an input from which the element identifier can be inherited. This applies only to non-collection output when mapping a collection over an input, since collections have a structured_like attribute. --- lib/galaxy/tools/execute.py | 13 ++++++++++++- lib/galaxy/tools/parser/xml.py | 1 + lib/galaxy/tools/parser/yaml.py | 1 + lib/galaxy/tools/xsd/galaxy.xsd | 6 ++++++ 4 files changed, 20 insertions(+), 1 deletion(-) diff --git a/lib/galaxy/tools/execute.py b/lib/galaxy/tools/execute.py index c926e5d09e1..578cffe930f 100644 --- a/lib/galaxy/tools/execute.py +++ b/lib/galaxy/tools/execute.py @@ -157,7 +157,18 @@ class ToolExecutionTracker( object ): log.warning( "Problem matching up datasets while attempting to create implicit dataset collections") continue output = self.tool.outputs[ output_name ] - element_identifiers = structure.element_identifiers_for_outputs( trans, outputs ) + + # Switch the structure for outputs if the output specified a default_identifier_source + collection_type_descriptions = trans.app.dataset_collections_service.collection_type_descriptions + _structure = None + source_collection = self.collection_info.collections.get(output.default_identifier_source) + if source_collection: + collection_type_description = collection_type_descriptions.for_collection_type(source_collection.collection.collection_type) + _structure = structure.for_dataset_collection( source_collection.collection, collection_type_description=collection_type_description) + if _structure and structure.can_match(_structure): + element_identifiers = _structure.element_identifiers_for_outputs(trans, outputs) + else: + element_identifiers = structure.element_identifiers_for_outputs( trans, outputs ) implicit_collection_info = dict( implicit_inputs=implicit_inputs, diff --git a/lib/galaxy/tools/parser/xml.py b/lib/galaxy/tools/parser/xml.py index 00e9d0eee08..cad49bb39a6 100644 --- a/lib/galaxy/tools/parser/xml.py +++ b/lib/galaxy/tools/parser/xml.py @@ -289,6 +289,7 @@ class XmlToolSource(ToolSource): output.format = output_format output.change_format = data_elem.findall("change_format") output.format_source = data_elem.get("format_source", default_format_source) + output.default_identifier_source = data_elem.get("default_identifier_source", 'None') output.metadata_source = data_elem.get("metadata_source", default_metadata_source) output.parent = data_elem.get("parent", None) output.label = xml_text( data_elem, "label" ) diff --git a/lib/galaxy/tools/parser/yaml.py b/lib/galaxy/tools/parser/yaml.py index 4450fb8f6d0..d45e9f4e478 100644 --- a/lib/galaxy/tools/parser/yaml.py +++ b/lib/galaxy/tools/parser/yaml.py @@ -113,6 +113,7 @@ class YamlToolSource(ToolSource): output.format = output_dict.get("format", "data") output.change_format = [] output.format_source = output_dict.get("format_source", None) + output.default_identifier_source = output_dict.get("default_identifier_source", None) output.metadata_source = output_dict.get("metadata_source", "") output.parent = output_dict.get("parent", None) output.label = output_dict.get( "label", None ) diff --git a/lib/galaxy/tools/xsd/galaxy.xsd b/lib/galaxy/tools/xsd/galaxy.xsd index 3d7bc3ef7a1..0abeed028e0 100644 --- a/lib/galaxy/tools/xsd/galaxy.xsd +++ b/lib/galaxy/tools/xsd/galaxy.xsd @@ -3693,6 +3693,12 @@ The valid values for format can be found in This sets the data type of the output file to be the same format as that of a tool input dataset. + + + Sets the source of element identifier to the specified input. +This only applies to collections that are mapped over a non-collection input and that have equivalent structures. + + This copies the metadata information From 35c0a4770336ca82fda7dd3034fd029e3bd474b8 Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Wed, 2 Aug 2017 15:22:15 +0200 Subject: [PATCH 2/4] Add test API test for default_identifier_source --- test/api/test_tools.py | 14 ++++++++++++++ test/functional/tools/samples_tool_conf.xml | 1 + 2 files changed, 15 insertions(+) diff --git a/test/api/test_tools.py b/test/api/test_tools.py index 4605ed7883e..e40b68b9b3f 100644 --- a/test/api/test_tools.py +++ b/test/api/test_tools.py @@ -1059,6 +1059,20 @@ class ToolsTestCase( api.ApiTestCase ): self.assertEquals( len( response_object[ 'jobs' ] ), 2 ) self.assertEquals( len( response_object[ 'implicit_collections' ] ), 1 ) + @skip_without_tool( "identifier_source" ) + def test_default_identifier_source_map_over(self): + history_id = self.dataset_populator.new_history() + input_a_hdca_id = self.dataset_collection_populator.create_list_in_history( history_id, contents=[("A", "A content")]).json()['id'] + input_b_hdca_id = self.dataset_collection_populator.create_list_in_history(history_id, contents=[("B", "B content")]).json()['id'] + inputs = { + "inputA": { 'batch': True, 'values': [ dict( src="hdca", id=input_a_hdca_id ) ] }, + "inputB": { 'batch': True, 'values': [ dict( src="hdca", id=input_b_hdca_id ) ] }, + } + self.dataset_populator.wait_for_history( history_id, assert_ok=True ) + create = self._run("identifier_source", history_id, inputs, assert_ok=True) + assert create['implicit_collections'][0]['elements'][0]['element_identifier'] == 'B' + assert create['implicit_collections'][1]['elements'][0]['element_identifier'] == 'A' + @skip_without_tool( "collection_creates_pair" ) def test_map_over_collection_output( self ): history_id = self.dataset_populator.new_history() diff --git a/test/functional/tools/samples_tool_conf.xml b/test/functional/tools/samples_tool_conf.xml index 4197eb61b5f..c4285378ca5 100644 --- a/test/functional/tools/samples_tool_conf.xml +++ b/test/functional/tools/samples_tool_conf.xml @@ -78,6 +78,7 @@ + From 46fba6ec1c728125efa1339e7334202487a09e18 Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Wed, 2 Aug 2017 18:09:30 +0200 Subject: [PATCH 3/4] Sort input keys by default, so the first matching input key will the source of the element identifiers --- lib/galaxy/dataset_collections/matching.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/galaxy/dataset_collections/matching.py b/lib/galaxy/dataset_collections/matching.py index 7230916a5d4..cccf08661aa 100644 --- a/lib/galaxy/dataset_collections/matching.py +++ b/lib/galaxy/dataset_collections/matching.py @@ -75,7 +75,7 @@ class MatchingCollections( object ): return None matching_collections = MatchingCollections() - for input_key, to_match in collections_to_match.items(): + for input_key, to_match in sorted(collections_to_match.items()): hdca = to_match.hdca collection_type_description = collection_type_descriptions.for_collection_type( hdca.collection.collection_type ) subcollection_type = to_match.subcollection_type From ea15f398e79771b6dd6ba5fdf0c8d96e10d39734 Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Wed, 2 Aug 2017 19:58:37 +0200 Subject: [PATCH 4/4] Fix failing API tests for explicit collection outputs --- lib/galaxy/tools/execute.py | 23 +++++++++++++---------- 1 file changed, 13 insertions(+), 10 deletions(-) diff --git a/lib/galaxy/tools/execute.py b/lib/galaxy/tools/execute.py index 578cffe930f..0751555133e 100644 --- a/lib/galaxy/tools/execute.py +++ b/lib/galaxy/tools/execute.py @@ -158,16 +158,19 @@ class ToolExecutionTracker( object ): continue output = self.tool.outputs[ output_name ] - # Switch the structure for outputs if the output specified a default_identifier_source - collection_type_descriptions = trans.app.dataset_collections_service.collection_type_descriptions - _structure = None - source_collection = self.collection_info.collections.get(output.default_identifier_source) - if source_collection: - collection_type_description = collection_type_descriptions.for_collection_type(source_collection.collection.collection_type) - _structure = structure.for_dataset_collection( source_collection.collection, collection_type_description=collection_type_description) - if _structure and structure.can_match(_structure): - element_identifiers = _structure.element_identifiers_for_outputs(trans, outputs) - else: + element_identifiers = None + if hasattr(output, "default_identifier_source"): + # Switch the structure for outputs if the output specified a default_identifier_source + collection_type_descriptions = trans.app.dataset_collections_service.collection_type_descriptions + + source_collection = self.collection_info.collections.get(output.default_identifier_source) + if source_collection: + collection_type_description = collection_type_descriptions.for_collection_type(source_collection.collection.collection_type) + _structure = structure.for_dataset_collection( source_collection.collection, collection_type_description=collection_type_description) + if structure.can_match(_structure): + element_identifiers = _structure.element_identifiers_for_outputs(trans, outputs) + + if not element_identifiers: element_identifiers = structure.element_identifiers_for_outputs( trans, outputs ) implicit_collection_info = dict(