diff --git a/lib/galaxy/managers/collections.py b/lib/galaxy/managers/collections.py index 2440bc843be..6f238185ef8 100644 --- a/lib/galaxy/managers/collections.py +++ b/lib/galaxy/managers/collections.py @@ -50,11 +50,13 @@ class DatasetCollectionManager( object ): element_identifiers=None, elements=None, implicit_collection_info=None, + trusted_identifiers=None, # Trust preloaded element objects ): """ """ # Trust embedded, newly created objects created by tool subsystem. - trusted_identifiers = implicit_collection_info is not None + if trusted_identifiers is None: + trusted_identifiers = implicit_collection_info is not None if element_identifiers and not trusted_identifiers: validate_input_element_identifiers( element_identifiers ) diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index 171c9252761..3472881beee 100755 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -343,16 +343,15 @@ class ToolOutputCollection( ToolOutputBase ): # This line is probably not right - should verify structured_like # or have outputs and all outputs have name. if len( self.outputs ) > 1: - outputs = self.outputs + output_parts = map( to_part, self.outputs ) else: # either must have specified structured_like or something worse if self.structure.structured_like: collection_prototype = inputs[ self.structure.structured_like ].collection else: collection_prototype = type_registry.prototype( self.structure.collection_type ) - # TODO: Handle nested structures. - outputs = odict() - for element in collection_prototype.elements: + + def prototype_dataset_element_to_output( element, parent_ids=[] ): name = element.element_identifier format = self.default_format if self.inherit_format: @@ -366,10 +365,29 @@ class ToolOutputCollection( ToolOutputBase ): ) if self.inherit_metadata: output.metadata_source = element.dataset_instance + return ToolOutputCollectionPart( + self, + element.element_identifier, + output, + parent_ids=parent_ids, + ) - outputs[ element.element_identifier ] = output + def prototype_collection_to_output( collection_prototype, parent_ids=[] ): + output_parts = [] + for element in collection_prototype.elements: + element_parts = [] + if not element.is_collection: + element_parts.append(prototype_dataset_element_to_output( element, parent_ids )) + else: + new_parent_ids = parent_ids[:] + [element.element_identifier] + element_parts.extend(prototype_collection_to_output(element.element_object, new_parent_ids)) + output_parts.extend(element_parts) - return map( to_part, outputs.items() ) + return output_parts + + output_parts = prototype_collection_to_output( collection_prototype ) + + return output_parts @property def dynamic_structure(self): @@ -402,10 +420,11 @@ class ToolOutputCollectionStructure( object ): class ToolOutputCollectionPart( object ): - def __init__( self, output_collection_def, element_identifier, output_def ): + def __init__( self, output_collection_def, element_identifier, output_def, parent_ids=[] ): self.output_collection_def = output_collection_def self.element_identifier = element_identifier self.output_def = output_def + self.parent_ids = parent_ids @property def effective_output_name( self ): diff --git a/lib/galaxy/tools/actions/__init__.py b/lib/galaxy/tools/actions/__init__.py index 93af340d2bd..cd42250db5b 100644 --- a/lib/galaxy/tools/actions/__init__.py +++ b/lib/galaxy/tools/actions/__init__.py @@ -287,16 +287,36 @@ class DefaultToolAction( object ): if not filter_output(output, incoming): if output.collection: collections_manager = trans.app.dataset_collections_service - # As far as I can tell - this is always true - but just verify assert set_output_history, "Cannot create dataset collection for this kind of tool." - elements = odict() + element_identifiers = [] input_collections = dict( [ (k, v[0]) for k, v in inp_dataset_collections.iteritems() ] ) known_outputs = output.known_outputs( input_collections, collections_manager.type_registry ) # Just to echo TODO elsewhere - this should be restructured to allow # nested collections. for output_part_def in known_outputs: + # Add elements to top-level collection, unless nested... + current_element_identifiers = element_identifiers + current_collection_type = output.structure.collection_type + + for parent_id in (output_part_def.parent_ids or []): + # TODO: replace following line with formal abstractions for doing this. + current_collection_type = ":".join(current_collection_type.split(":")[1:]) + name_to_index = dict(map(lambda (index, value): (value["name"], index), enumerate(current_element_identifiers))) + if parent_id not in name_to_index: + if parent_id not in current_element_identifiers: + index = len(current_element_identifiers) + current_element_identifiers.append(dict( + name=parent_id, + collection_type=current_collection_type, + src="new_collection", + element_identifiers=[], + )) + else: + index = name_to_index[parent_id] + current_element_identifiers = current_element_identifiers[ index ][ "element_identifiers" ] + effective_output_name = output_part_def.effective_output_name element = handle_output( effective_output_name, output_part_def.output_def ) # Following hack causes dataset to no be added to history... @@ -307,17 +327,23 @@ class DefaultToolAction( object ): trans.sa_session.add( element ) trans.sa_session.flush() - elements[ output_part_def.element_identifier ] = element + current_element_identifiers.append({ + "__object__": element, + "name": output_part_def.element_identifier, + }) + log.info(element_identifiers) if output.dynamic_structure: - assert not elements # known_outputs must have been empty - elements = collections_manager.ELEMENTS_UNINITIALIZED + assert not element_identifiers # known_outputs must have been empty + element_kwds = dict(elements=collections_manager.ELEMENTS_UNINITIALIZED) + else: + element_kwds = dict(element_identifiers=element_identifiers) if mapping_over_collection: dc = collections_manager.create_dataset_collection( trans, collection_type=output.structure.collection_type, - elements=elements, + **element_kwds ) out_collections[ name ] = dc else: @@ -327,7 +353,8 @@ class DefaultToolAction( object ): history, name=hdca_name, collection_type=output.structure.collection_type, - elements=elements, + trusted_identifiers=True, + **element_kwds ) # name here is name of the output element - not name # of the hdca. diff --git a/lib/galaxy/tools/parser/xml.py b/lib/galaxy/tools/parser/xml.py index 56667d4242c..0a03d387742 100644 --- a/lib/galaxy/tools/parser/xml.py +++ b/lib/galaxy/tools/parser/xml.py @@ -359,17 +359,22 @@ def __parse_output_collection_elem( output_collection_elem ): name = attrib.pop( 'name', None ) if name is None: raise Exception( "Test output collection does not have a 'name'" ) + element_tests = __parse_element_tests( output_collection_elem ) + return TestCollectionOutputDef( name, attrib, element_tests ) + + +def __parse_element_tests( parent_element ): element_tests = {} - for element in output_collection_elem.findall("element"): + for element in parent_element.findall("element"): element_attrib = dict( element.attrib ) identifier = element_attrib.pop( 'name', None ) if identifier is None: raise Exception( "Test primary dataset does not have a 'identifier'" ) - element_tests[ identifier ] = __parse_test_attributes( element, element_attrib ) - return TestCollectionOutputDef( name, attrib, element_tests ) + element_tests[ identifier ] = __parse_test_attributes( element, element_attrib, parse_elements=True ) + return element_tests -def __parse_test_attributes( output_elem, attrib ): +def __parse_test_attributes( output_elem, attrib, parse_elements=False ): assert_list = __parse_assert_list( output_elem ) file = attrib.pop( 'file', None ) # File no longer required if an list of assertions was present. @@ -390,12 +395,17 @@ def __parse_test_attributes( output_elem, attrib ): for metadata_elem in output_elem.findall( 'metadata' ): metadata[ metadata_elem.get('name') ] = metadata_elem.get( 'value' ) md5sum = attrib.get("md5", None) - if not (assert_list or file or extra_files or metadata or md5sum): + element_tests = {} + if parse_elements: + element_tests = __parse_element_tests( output_elem ) + + if not (assert_list or file or extra_files or metadata or md5sum or element_tests): raise Exception( "Test output defines nothing to check (e.g. must have a 'file' check against, assertions to check, metadata or md5 tests, etc...)") attributes['assert_list'] = assert_list attributes['extra_files'] = extra_files attributes['metadata'] = metadata attributes['md5'] = md5sum + attributes['elements'] = element_tests return file, attributes diff --git a/test/functional/test_toolbox.py b/test/functional/test_toolbox.py index c74660a4daa..16e8602f903 100644 --- a/test/functional/test_toolbox.py +++ b/test/functional/test_toolbox.py @@ -191,8 +191,12 @@ class ToolTestCase( TwillTestCase ): # the job completed so re-hit the API for more information. data_collection_returned = data_collection_list[ name ] data_collection = galaxy_interactor._get( "dataset_collections/%s" % data_collection_returned[ "id" ], data={"instance_type": "history"} ).json() - elements = data_collection[ "elements" ] - element_dict = dict( map(lambda e: (e["element_identifier"], e["object"]), elements) ) + + def get_element( elements, id ): + for element in elements: + if element["element_identifier"] == id: + return element + return False expected_collection_type = output_collection_def.collection_type if expected_collection_type: @@ -202,20 +206,29 @@ class ToolTestCase( TwillTestCase ): message = template % (name, expected_collection_type, collection_type) raise AssertionError(message) - for element_identifier, ( element_outfile, element_attrib ) in output_collection_def.element_tests.items(): - if element_identifier not in element_dict: - template = "Failed to find identifier [%s] for testing, tool generated collection with identifiers [%s]" - message = template % (element_identifier, ",".join(element_dict.keys())) - raise AssertionError(message) - hda = element_dict[ element_identifier ] + def verify_elements( element_objects, element_tests ): + for element_identifier, ( element_outfile, element_attrib ) in element_tests.items(): + element = get_element( element_objects, element_identifier ) + if not element: + template = "Failed to find identifier [%s] for testing, tool generated collection elements [%s]" + message = template % (element_identifier, element_objects) + raise AssertionError(message) - galaxy_interactor.verify_output_dataset( - history, - hda_id=hda["id"], - outfile=element_outfile, - attributes=element_attrib, - shed_tool_id=shed_tool_id - ) + element_type = element["element_type"] + if element_type != "dataset_collection": + hda = element[ "object" ] + galaxy_interactor.verify_output_dataset( + history, + hda_id=hda["id"], + outfile=element_outfile, + attributes=element_attrib, + shed_tool_id=shed_tool_id + ) + if element_type == "dataset_collection": + elements = element[ "object" ][ "elements" ] + verify_elements( elements, element_attrib.get( "elements", {} ) ) + + verify_elements( data_collection[ "elements" ], output_collection_def.element_tests ) except Exception as e: register_exception(e) diff --git a/test/functional/tools/collection_creates_list_of_pairs.xml b/test/functional/tools/collection_creates_list_of_pairs.xml new file mode 100644 index 00000000000..3b7b7804162 --- /dev/null +++ b/test/functional/tools/collection_creates_list_of_pairs.xml @@ -0,0 +1,50 @@ + + + + #for $list_key in $list_output.keys()# + #for $pair_key in $list_output[$list_key].keys()# + echo "identifier is $list_key:$pair_key" > "$list_output[$list_key][$pair_key]"; + #end for# + #end for# + echo 'ensure not empty'; + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/test/functional/tools/samples_tool_conf.xml b/test/functional/tools/samples_tool_conf.xml index 4b1f5fb7001..73a4708c9d2 100644 --- a/test/functional/tools/samples_tool_conf.xml +++ b/test/functional/tools/samples_tool_conf.xml @@ -55,6 +55,7 @@ +