Merge pull request #4368 from mvdbeek/tie_down_collection_matcher

Implement default_identifier_source for outputs and sort input keys by default
This commit is contained in:
John Chilton
2017-08-07 07:45:37 -04:00
committed by GitHub
7 changed files with 39 additions and 2 deletions
+1 -1
View File
@@ -75,7 +75,7 @@ class MatchingCollections( object ):
return None
matching_collections = MatchingCollections()
for input_key, to_match in collections_to_match.items():
for input_key, to_match in sorted(collections_to_match.items()):
hdca = to_match.hdca
collection_type_description = collection_type_descriptions.for_collection_type( hdca.collection.collection_type )
subcollection_type = to_match.subcollection_type
+15 -1
View File
@@ -157,7 +157,21 @@ class ToolExecutionTracker( object ):
log.warning( "Problem matching up datasets while attempting to create implicit dataset collections")
continue
output = self.tool.outputs[ output_name ]
element_identifiers = structure.element_identifiers_for_outputs( trans, outputs )
element_identifiers = None
if hasattr(output, "default_identifier_source"):
# Switch the structure for outputs if the output specified a default_identifier_source
collection_type_descriptions = trans.app.dataset_collections_service.collection_type_descriptions
source_collection = self.collection_info.collections.get(output.default_identifier_source)
if source_collection:
collection_type_description = collection_type_descriptions.for_collection_type(source_collection.collection.collection_type)
_structure = structure.for_dataset_collection( source_collection.collection, collection_type_description=collection_type_description)
if structure.can_match(_structure):
element_identifiers = _structure.element_identifiers_for_outputs(trans, outputs)
if not element_identifiers:
element_identifiers = structure.element_identifiers_for_outputs( trans, outputs )
implicit_collection_info = dict(
implicit_inputs=implicit_inputs,
+1
View File
@@ -289,6 +289,7 @@ class XmlToolSource(ToolSource):
output.format = output_format
output.change_format = data_elem.findall("change_format")
output.format_source = data_elem.get("format_source", default_format_source)
output.default_identifier_source = data_elem.get("default_identifier_source", 'None')
output.metadata_source = data_elem.get("metadata_source", default_metadata_source)
output.parent = data_elem.get("parent", None)
output.label = xml_text( data_elem, "label" )
+1
View File
@@ -113,6 +113,7 @@ class YamlToolSource(ToolSource):
output.format = output_dict.get("format", "data")
output.change_format = []
output.format_source = output_dict.get("format_source", None)
output.default_identifier_source = output_dict.get("default_identifier_source", None)
output.metadata_source = output_dict.get("metadata_source", "")
output.parent = output_dict.get("parent", None)
output.label = output_dict.get( "label", None )
+6
View File
@@ -3693,6 +3693,12 @@ The valid values for format can be found in
<xs:documentation xml:lang="en">This sets the data type of the output file to be the same format as that of a tool input dataset.</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="default_identifier_source" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">Sets the source of element identifier to the specified input.
This only applies to collections that are mapped over a non-collection input and that have equivalent structures.</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="metadata_source" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">This copies the metadata information
+14
View File
@@ -1059,6 +1059,20 @@ class ToolsTestCase( api.ApiTestCase ):
self.assertEquals( len( response_object[ 'jobs' ] ), 2 )
self.assertEquals( len( response_object[ 'implicit_collections' ] ), 1 )
@skip_without_tool( "identifier_source" )
def test_default_identifier_source_map_over(self):
history_id = self.dataset_populator.new_history()
input_a_hdca_id = self.dataset_collection_populator.create_list_in_history( history_id, contents=[("A", "A content")]).json()['id']
input_b_hdca_id = self.dataset_collection_populator.create_list_in_history(history_id, contents=[("B", "B content")]).json()['id']
inputs = {
"inputA": { 'batch': True, 'values': [ dict( src="hdca", id=input_a_hdca_id ) ] },
"inputB": { 'batch': True, 'values': [ dict( src="hdca", id=input_b_hdca_id ) ] },
}
self.dataset_populator.wait_for_history( history_id, assert_ok=True )
create = self._run("identifier_source", history_id, inputs, assert_ok=True)
assert create['implicit_collections'][0]['elements'][0]['element_identifier'] == 'B'
assert create['implicit_collections'][1]['elements'][0]['element_identifier'] == 'A'
@skip_without_tool( "collection_creates_pair" )
def test_map_over_collection_output( self ):
history_id = self.dataset_populator.new_history()
@@ -78,6 +78,7 @@
<tool file="validation_empty_dataset.xml" />
<tool file="implicit_conversion.xml" />
<tool file="explicit_conversion.xml" />
<tool file="identifier_source.xml" />
<tool file="identifier_single.xml" />
<tool file="identifier_multiple.xml" />
<tool file="identifier_multiple_in_conditional.xml" />