Allow tools to explicitly create nested collections with static structure.

https://bitbucket.org/galaxy/galaxy-central/pull-request/634/allow-tools-to-explicitly-produce-dataset added the ability for tool to create simple collections (lists and pairs). It also outlined three creation scenarios - fixed collections, pre-determinable collection structures (like output lists based on input lists), and fully dynamic output collections. This pull request allows tool to output nested collections for these first two.

Fully dynamic nested collections (using <discover_datasets> tags) is not implemented in this commit.

Additionally, the tool test syntax has been extended to allow testing nested collection elements and an example tool is included that demonstrates this functionality - test/functional/tools/collection_creates_list_of_pairs.xml.
This commit is contained in:
John Chilton
2015-07-31 14:10:54 +01:00
parent be3bd7a296
commit 13ea86753a
7 changed files with 157 additions and 35 deletions
+3 -1
View File
@@ -50,11 +50,13 @@ class DatasetCollectionManager( object ):
element_identifiers=None,
elements=None,
implicit_collection_info=None,
trusted_identifiers=None, # Trust preloaded element objects
):
"""
"""
# Trust embedded, newly created objects created by tool subsystem.
trusted_identifiers = implicit_collection_info is not None
if trusted_identifiers is None:
trusted_identifiers = implicit_collection_info is not None
if element_identifiers and not trusted_identifiers:
validate_input_element_identifiers( element_identifiers )
+26 -7
View File
@@ -343,16 +343,15 @@ class ToolOutputCollection( ToolOutputBase ):
# This line is probably not right - should verify structured_like
# or have outputs and all outputs have name.
if len( self.outputs ) > 1:
outputs = self.outputs
output_parts = map( to_part, self.outputs )
else:
# either must have specified structured_like or something worse
if self.structure.structured_like:
collection_prototype = inputs[ self.structure.structured_like ].collection
else:
collection_prototype = type_registry.prototype( self.structure.collection_type )
# TODO: Handle nested structures.
outputs = odict()
for element in collection_prototype.elements:
def prototype_dataset_element_to_output( element, parent_ids=[] ):
name = element.element_identifier
format = self.default_format
if self.inherit_format:
@@ -366,10 +365,29 @@ class ToolOutputCollection( ToolOutputBase ):
)
if self.inherit_metadata:
output.metadata_source = element.dataset_instance
return ToolOutputCollectionPart(
self,
element.element_identifier,
output,
parent_ids=parent_ids,
)
outputs[ element.element_identifier ] = output
def prototype_collection_to_output( collection_prototype, parent_ids=[] ):
output_parts = []
for element in collection_prototype.elements:
element_parts = []
if not element.is_collection:
element_parts.append(prototype_dataset_element_to_output( element, parent_ids ))
else:
new_parent_ids = parent_ids[:] + [element.element_identifier]
element_parts.extend(prototype_collection_to_output(element.element_object, new_parent_ids))
output_parts.extend(element_parts)
return map( to_part, outputs.items() )
return output_parts
output_parts = prototype_collection_to_output( collection_prototype )
return output_parts
@property
def dynamic_structure(self):
@@ -402,10 +420,11 @@ class ToolOutputCollectionStructure( object ):
class ToolOutputCollectionPart( object ):
def __init__( self, output_collection_def, element_identifier, output_def ):
def __init__( self, output_collection_def, element_identifier, output_def, parent_ids=[] ):
self.output_collection_def = output_collection_def
self.element_identifier = element_identifier
self.output_def = output_def
self.parent_ids = parent_ids
@property
def effective_output_name( self ):
+34 -7
View File
@@ -287,16 +287,36 @@ class DefaultToolAction( object ):
if not filter_output(output, incoming):
if output.collection:
collections_manager = trans.app.dataset_collections_service
# As far as I can tell - this is always true - but just verify
assert set_output_history, "Cannot create dataset collection for this kind of tool."
elements = odict()
element_identifiers = []
input_collections = dict( [ (k, v[0]) for k, v in inp_dataset_collections.iteritems() ] )
known_outputs = output.known_outputs( input_collections, collections_manager.type_registry )
# Just to echo TODO elsewhere - this should be restructured to allow
# nested collections.
for output_part_def in known_outputs:
# Add elements to top-level collection, unless nested...
current_element_identifiers = element_identifiers
current_collection_type = output.structure.collection_type
for parent_id in (output_part_def.parent_ids or []):
# TODO: replace following line with formal abstractions for doing this.
current_collection_type = ":".join(current_collection_type.split(":")[1:])
name_to_index = dict(map(lambda (index, value): (value["name"], index), enumerate(current_element_identifiers)))
if parent_id not in name_to_index:
if parent_id not in current_element_identifiers:
index = len(current_element_identifiers)
current_element_identifiers.append(dict(
name=parent_id,
collection_type=current_collection_type,
src="new_collection",
element_identifiers=[],
))
else:
index = name_to_index[parent_id]
current_element_identifiers = current_element_identifiers[ index ][ "element_identifiers" ]
effective_output_name = output_part_def.effective_output_name
element = handle_output( effective_output_name, output_part_def.output_def )
# Following hack causes dataset to no be added to history...
@@ -307,17 +327,23 @@ class DefaultToolAction( object ):
trans.sa_session.add( element )
trans.sa_session.flush()
elements[ output_part_def.element_identifier ] = element
current_element_identifiers.append({
"__object__": element,
"name": output_part_def.element_identifier,
})
log.info(element_identifiers)
if output.dynamic_structure:
assert not elements # known_outputs must have been empty
elements = collections_manager.ELEMENTS_UNINITIALIZED
assert not element_identifiers # known_outputs must have been empty
element_kwds = dict(elements=collections_manager.ELEMENTS_UNINITIALIZED)
else:
element_kwds = dict(element_identifiers=element_identifiers)
if mapping_over_collection:
dc = collections_manager.create_dataset_collection(
trans,
collection_type=output.structure.collection_type,
elements=elements,
**element_kwds
)
out_collections[ name ] = dc
else:
@@ -327,7 +353,8 @@ class DefaultToolAction( object ):
history,
name=hdca_name,
collection_type=output.structure.collection_type,
elements=elements,
trusted_identifiers=True,
**element_kwds
)
# name here is name of the output element - not name
# of the hdca.
+15 -5
View File
@@ -359,17 +359,22 @@ def __parse_output_collection_elem( output_collection_elem ):
name = attrib.pop( 'name', None )
if name is None:
raise Exception( "Test output collection does not have a 'name'" )
element_tests = __parse_element_tests( output_collection_elem )
return TestCollectionOutputDef( name, attrib, element_tests )
def __parse_element_tests( parent_element ):
element_tests = {}
for element in output_collection_elem.findall("element"):
for element in parent_element.findall("element"):
element_attrib = dict( element.attrib )
identifier = element_attrib.pop( 'name', None )
if identifier is None:
raise Exception( "Test primary dataset does not have a 'identifier'" )
element_tests[ identifier ] = __parse_test_attributes( element, element_attrib )
return TestCollectionOutputDef( name, attrib, element_tests )
element_tests[ identifier ] = __parse_test_attributes( element, element_attrib, parse_elements=True )
return element_tests
def __parse_test_attributes( output_elem, attrib ):
def __parse_test_attributes( output_elem, attrib, parse_elements=False ):
assert_list = __parse_assert_list( output_elem )
file = attrib.pop( 'file', None )
# File no longer required if an list of assertions was present.
@@ -390,12 +395,17 @@ def __parse_test_attributes( output_elem, attrib ):
for metadata_elem in output_elem.findall( 'metadata' ):
metadata[ metadata_elem.get('name') ] = metadata_elem.get( 'value' )
md5sum = attrib.get("md5", None)
if not (assert_list or file or extra_files or metadata or md5sum):
element_tests = {}
if parse_elements:
element_tests = __parse_element_tests( output_elem )
if not (assert_list or file or extra_files or metadata or md5sum or element_tests):
raise Exception( "Test output defines nothing to check (e.g. must have a 'file' check against, assertions to check, metadata or md5 tests, etc...)")
attributes['assert_list'] = assert_list
attributes['extra_files'] = extra_files
attributes['metadata'] = metadata
attributes['md5'] = md5sum
attributes['elements'] = element_tests
return file, attributes
+28 -15
View File
@@ -191,8 +191,12 @@ class ToolTestCase( TwillTestCase ):
# the job completed so re-hit the API for more information.
data_collection_returned = data_collection_list[ name ]
data_collection = galaxy_interactor._get( "dataset_collections/%s" % data_collection_returned[ "id" ], data={"instance_type": "history"} ).json()
elements = data_collection[ "elements" ]
element_dict = dict( map(lambda e: (e["element_identifier"], e["object"]), elements) )
def get_element( elements, id ):
for element in elements:
if element["element_identifier"] == id:
return element
return False
expected_collection_type = output_collection_def.collection_type
if expected_collection_type:
@@ -202,20 +206,29 @@ class ToolTestCase( TwillTestCase ):
message = template % (name, expected_collection_type, collection_type)
raise AssertionError(message)
for element_identifier, ( element_outfile, element_attrib ) in output_collection_def.element_tests.items():
if element_identifier not in element_dict:
template = "Failed to find identifier [%s] for testing, tool generated collection with identifiers [%s]"
message = template % (element_identifier, ",".join(element_dict.keys()))
raise AssertionError(message)
hda = element_dict[ element_identifier ]
def verify_elements( element_objects, element_tests ):
for element_identifier, ( element_outfile, element_attrib ) in element_tests.items():
element = get_element( element_objects, element_identifier )
if not element:
template = "Failed to find identifier [%s] for testing, tool generated collection elements [%s]"
message = template % (element_identifier, element_objects)
raise AssertionError(message)
galaxy_interactor.verify_output_dataset(
history,
hda_id=hda["id"],
outfile=element_outfile,
attributes=element_attrib,
shed_tool_id=shed_tool_id
)
element_type = element["element_type"]
if element_type != "dataset_collection":
hda = element[ "object" ]
galaxy_interactor.verify_output_dataset(
history,
hda_id=hda["id"],
outfile=element_outfile,
attributes=element_attrib,
shed_tool_id=shed_tool_id
)
if element_type == "dataset_collection":
elements = element[ "object" ][ "elements" ]
verify_elements( elements, element_attrib.get( "elements", {} ) )
verify_elements( data_collection[ "elements" ], output_collection_def.element_tests )
except Exception as e:
register_exception(e)
@@ -0,0 +1,50 @@
<tool id="collection_creates_list_of_pairs" name="collection_creates_list_or_pairs" version="0.1.0">
<!-- You usually wouldn't want to do this - just write the operation for
a single dataset and allow the user to map that tool over the whole
collection. -->
<command>
#for $list_key in $list_output.keys()#
#for $pair_key in $list_output[$list_key].keys()#
echo "identifier is $list_key:$pair_key" > "$list_output[$list_key][$pair_key]";
#end for#
#end for#
echo 'ensure not empty';
</command>
<inputs>
<param name="input1" type="data_collection" collection_type="list:paired" label="Input" help="Input collection..." format="txt" />
</inputs>
<outputs>
<collection name="list_output" type="list:paired" label="Duplicate List" structured_like="input1" inherit_format="true">
<!-- inherit_format can be used in conjunction with structured_like
to perserve format. -->
</collection>
</outputs>
<tests>
<test>
<param name="input1">
<collection type="list:paired">
<element name="i1">
<collection type="paired">
<element name="forward" value="simple_line.txt" />
<element name="reverse" value="simple_line_alternative.txt" />
</collection>
</element>
</collection>
</param>
<output_collection name="list_output" type="list:paired">
<element name="i1">
<element name="forward">
<assert_contents>
<has_text_matching expression="^identifier is i1:forward\n$" />
</assert_contents>
</element>
<element name="reverse">
<assert_contents>
<has_text_matching expression="^identifier is i1:reverse\n$" />
</assert_contents>
</element>
</element>
</output_collection>
</test>
</tests>
</tool>
@@ -55,6 +55,7 @@
<tool file="collection_creates_pair_from_type.xml" />
<tool file="collection_creates_list.xml" />
<tool file="collection_creates_list_2.xml" />
<tool file="collection_creates_list_of_pairs.xml" />
<tool file="collection_optional_param.xml" />
<tool file="collection_split_on_column.xml" />