More configurable format and metadata handling for output collections.

Imporvements to testing code.
This commit is contained in:
John Chilton
2015-01-15 09:30:00 -05:00
parent 4c5c8a47db
commit 2cb7c8d73e
10 changed files with 179 additions and 64 deletions
+31 -4
View File
@@ -290,15 +290,30 @@ class ToolOutputCollection( ToolOutputBase ):
<outputs>
"""
def __init__( self, name, structure, label=None, filters=None, hidden=False, default_format="data" ):
def __init__(
self,
name,
structure,
label=None,
filters=None,
hidden=False,
default_format="data",
default_format_source=None,
default_metadata_source=None,
inherit_format=False,
inherit_metadata=False
):
super( ToolOutputCollection, self ).__init__( name, label=label, filters=filters, hidden=hidden )
self.collection = True
self.default_format = default_format
self.structure = structure
self.outputs = odict()
# TODO:
self.metadata_source = None
self.inherit_format = inherit_format
self.inherit_metadata = inherit_metadata
self.metadata_source = default_metadata_source
self.format_source = default_format_source
def known_outputs( self, inputs ):
if self.dynamic_structure:
@@ -317,7 +332,19 @@ class ToolOutputCollection( ToolOutputBase ):
outputs = odict()
for element in input_collection.collection.elements:
name = element.element_identifier
output = ToolOutput( name, format=self.default_format, implicit=True )
format = self.default_format
if self.inherit_format:
format = element.dataset_instance.ext
output = ToolOutput(
name,
format=format,
format_source=self.format_source,
metadata_source=self.metadata_source,
implicit=True,
)
if self.inherit_metadata:
output.metadata_source = element.dataset_instance
outputs[ element.element_identifier ] = output
return map( to_part, outputs.items() )
+15 -2
View File
@@ -245,8 +245,16 @@ class DefaultToolAction( object ):
# This may not be neccesary with the new parent/child associations
data.designation = name
# Copy metadata from one of the inputs if requested.
if output.metadata_source:
data.init_meta( copy_from=inp_data[output.metadata_source] )
# metadata source can be either a string referencing an input
# or an actual object to copy.
metadata_source = output.metadata_source
if metadata_source:
if isinstance( metadata_source, basestring ):
metadata_source = inp_data[metadata_source]
if metadata_source is not None:
data.init_meta( copy_from=metadata_source )
else:
data.init_meta()
# Take dbkey from LAST input
@@ -287,6 +295,11 @@ class DefaultToolAction( object ):
# Following hack causes dataset to no be added to history...
child_dataset_names.add( effective_output_name )
if set_output_history:
history.add_dataset( element, set_hid=set_output_hid )
trans.sa_session.add( element )
trans.sa_session.flush()
elements[ output_part_def.element_identifier ] = element
if output.dynamic_structure:
@@ -174,6 +174,8 @@ class JobContext( object ):
# Associate new dataset with job
if self.job:
self.job.history.add_dataset( primary_data )
assoc = app.model.JobToOutputDatasetAssociation( '__new_primary_file_%s|%s__' % ( name, designation ), primary_data )
assoc.job = self.job
sa_session.add( assoc )
+31 -7
View File
@@ -141,8 +141,8 @@ class XmlToolSource(ToolSource):
data_dict = odict()
def _parse(data_elem, default_format="data"):
output_def = self._parse_output(data_elem, tool, default_format=default_format)
def _parse(data_elem, **kwds):
output_def = self._parse_output(data_elem, tool, **kwds)
data_dict[output_def.name] = output_def
return output_def
@@ -153,6 +153,14 @@ class XmlToolSource(ToolSource):
default_format = collection_elem.get( "format", "data" )
collection_type = collection_elem.get( "type", None )
structured_like = collection_elem.get( "structured_like", None )
inherit_format = False
inherit_metadata = False
if structured_like:
inherit_format = string_as_bool( collection_elem.get( "inherit_format", None ) )
inherit_metadata = string_as_bool( collection_elem.get( "inherit_metadata", None ) )
default_format_source = collection_elem.get( "format_source", None )
default_metadata_source = collection_elem.get( "metadata_source", "" )
dataset_collectors = None
if collection_elem.find( "discover_datasets" ) is not None:
dataset_collectors = output_collect.dataset_collectors_from_elem( collection_elem )
@@ -164,12 +172,21 @@ class XmlToolSource(ToolSource):
output_collection = galaxy.tools.ToolOutputCollection(
name,
structure,
default_format=default_format
default_format=default_format,
inherit_format=inherit_format,
inherit_metadata=inherit_metadata,
default_format_source=default_format_source,
default_metadata_source=default_metadata_source,
)
outputs[output_collection.name] = output_collection
for data_elem in collection_elem.findall("data"):
_parse( data_elem, default_format=default_format )
_parse(
data_elem,
default_format=default_format,
default_format_source=default_format_source,
default_metadata_source=default_metadata_source,
)
for data_elem in collection_elem.findall("data"):
output_name = data_elem.get("name")
@@ -183,12 +200,19 @@ class XmlToolSource(ToolSource):
outputs[output_def.name] = output_def
return outputs, output_collections
def _parse_output(self, data_elem, tool, default_format):
def _parse_output(
self,
data_elem,
tool,
default_format="data",
default_format_source=None,
default_metadata_source="",
):
output = galaxy.tools.ToolOutput( data_elem.get("name") )
output.format = data_elem.get("format", default_format)
output.change_format = data_elem.findall("change_format")
output.format_source = data_elem.get("format_source", None)
output.metadata_source = data_elem.get("metadata_source", "")
output.format_source = data_elem.get("format_source", default_format_source)
output.metadata_source = data_elem.get("metadata_source", default_metadata_source)
output.parent = data_elem.get("parent", None)
output.label = xml_text( data_elem, "label" )
output.count = int( data_elem.get("count", 1) )
+1 -1
View File
@@ -134,7 +134,7 @@ class DatasetPopulator( object ):
def get_history_dataset_details( self, history_id, **kwds ):
dataset_id = self.__history_dataset_id( history_id, **kwds )
details_response = self.__get_contents_request( history_id, "/%s" % dataset_id )
details_response = self.__get_contents_request( history_id, "/datasets/%s" % dataset_id )
assert details_response.status_code == 200
return details_response.json()
+67 -47
View File
@@ -225,25 +225,11 @@ class ToolsTestCase( api.ApiTestCase ):
# TODO: shouldn't need this wait
self.dataset_populator.wait_for_history( history_id, assert_ok=True )
create = self._run( "collection_creates_pair", history_id, inputs, assert_ok=True )
jobs = create[ 'jobs' ]
implicit_collections = create[ 'implicit_collections' ]
collections = create[ 'output_collections' ]
self.assertEquals( len( jobs ), 1 )
self.assertEquals( len( implicit_collections ), 0 )
self.assertEquals( len( collections ), 1 )
output_collection = collections[ 0 ]
elements = output_collection[ "elements" ]
assert len( elements ) == 2
element0, element1 = elements
assert element0[ "element_identifier" ] == "forward"
assert element1[ "element_identifier" ] == "reverse"
output_collection = self._assert_one_job_one_collection_run( create )
element0, element1 = self._assert_elements_are( output_collection, "forward", "reverse" )
self.dataset_populator.wait_for_history( history_id, assert_ok=True )
contents0 = self.dataset_populator.get_history_dataset_content( history_id, dataset_id=element0["object"]["id"])
assert contents0 == "123\n789\n", contents0
contents1 = self.dataset_populator.get_history_dataset_content( history_id, dataset_id=element1["object"]["id"])
assert contents1 == "456\n0ab\n", contents1
self._verify_element( history_id, element0, contents="123\n789\n", file_ext="txt" )
self._verify_element( history_id, element1, contents="456\n0ab\n", file_ext="txt" )
@skip_without_tool( "collection_creates_list" )
def test_list_collection_output( self ):
@@ -256,25 +242,31 @@ class ToolsTestCase( api.ApiTestCase ):
# TODO: real problem here - shouldn't have to have this wait.
self.dataset_populator.wait_for_history( history_id, assert_ok=True )
create = self._run( "collection_creates_list", history_id, inputs, assert_ok=True )
jobs = create[ 'jobs' ]
implicit_collections = create[ 'implicit_collections' ]
collections = create[ 'output_collections' ]
self.assertEquals( len( jobs ), 1 )
self.assertEquals( len( implicit_collections ), 0 )
self.assertEquals( len( collections ), 1 )
output_collection = collections[ 0 ]
elements = output_collection[ "elements" ]
assert len( elements ) == 2
element0, element1 = elements
assert element0[ "element_identifier" ] == "data1"
assert element1[ "element_identifier" ] == "data2"
output_collection = self._assert_one_job_one_collection_run( create )
element0, element1 = self._assert_elements_are( output_collection, "data1", "data2" )
self.dataset_populator.wait_for_history( history_id, assert_ok=True )
contents0 = self.dataset_populator.get_history_dataset_content( history_id, dataset_id=element0["object"]["id"])
assert contents0 == "0\n", contents0
contents1 = self.dataset_populator.get_history_dataset_content( history_id, dataset_id=element1["object"]["id"])
assert contents1 == "1\n", contents1
self._verify_element( history_id, element0, contents="identifier is data1\n", file_ext="txt" )
self._verify_element( history_id, element1, contents="identifier is data2\n", file_ext="txt" )
@skip_without_tool( "collection_creates_list_2" )
def test_list_collection_output_format_source( self ):
# test using format_source with a tool
history_id = self.dataset_populator.new_history()
new_dataset1 = self.dataset_populator.new_dataset( history_id, content='#col1\tcol2' )
create_response = self.dataset_collection_populator.create_list_in_history( history_id, contents=["a\tb\nc\td", "e\tf\ng\th"] )
hdca_id = create_response.json()[ "id" ]
inputs = {
"header": { "src": "hda", "id": new_dataset1["id"] },
"input_collect": { "src": "hdca", "id": hdca_id },
}
# TODO: real problem here - shouldn't have to have this wait.
self.dataset_populator.wait_for_history( history_id, assert_ok=True )
create = self._run( "collection_creates_list_2", history_id, inputs, assert_ok=True )
output_collection = self._assert_one_job_one_collection_run( create )
element0, element1 = self._assert_elements_are( output_collection, "data1", "data2" )
self.dataset_populator.wait_for_history( history_id, assert_ok=True )
self._verify_element( history_id, element0, contents="#col1\tcol2\na\tb\nc\td\n", file_ext="txt" )
self._verify_element( history_id, element1, contents="#col1\tcol2\ne\tf\ng\th\n", file_ext="txt" )
@skip_without_tool( "collection_split_on_column" )
def test_dynamic_list_output( self ):
@@ -286,21 +278,12 @@ class ToolsTestCase( api.ApiTestCase ):
self.dataset_populator.wait_for_history( history_id, assert_ok=True )
create = self._run( "collection_split_on_column", history_id, inputs, assert_ok=True )
jobs = create[ 'jobs' ]
implicit_collections = create[ 'implicit_collections' ]
collections = create[ 'output_collections' ]
self.assertEquals( len( jobs ), 1 )
job_id = jobs[ 0 ][ "id" ]
self.assertEquals( len( implicit_collections ), 0 )
self.assertEquals( len( collections ), 1 )
output_collection = collections[0]
output_collection = self._assert_one_job_one_collection_run( create )
self._assert_has_keys( output_collection, "id", "name", "elements", "populated" )
assert not output_collection[ "populated" ]
assert len( output_collection[ "elements" ] ) == 0
self.dataset_populator.wait_for_job( job_id, assert_ok=True )
self.dataset_populator.wait_for_job( create["jobs"][0]["id"], assert_ok=True )
get_collection_response = self._get( "dataset_collections/%s" % output_collection[ "id" ], data={"instance_type": "history"} )
self._assert_status_code_is( get_collection_response, 200 )
@@ -309,6 +292,7 @@ class ToolsTestCase( api.ApiTestCase ):
self._assert_has_keys( output_collection, "id", "name", "elements", "populated" )
assert output_collection[ "populated" ]
assert len( output_collection[ "elements" ] ) == 2
# TODO: verify element identifiers
@skip_without_tool( "cat1" )
def test_run_cat1_with_two_inputs( self ):
@@ -443,6 +427,42 @@ class ToolsTestCase( api.ApiTestCase ):
assert "123\n0ab" in outputs_contents
assert "456\n789" in outputs_contents
def _assert_one_job_one_collection_run( self, create ):
jobs = create[ 'jobs' ]
implicit_collections = create[ 'implicit_collections' ]
collections = create[ 'output_collections' ]
self.assertEquals( len( jobs ), 1 )
self.assertEquals( len( implicit_collections ), 0 )
self.assertEquals( len( collections ), 1 )
output_collection = collections[ 0 ]
return output_collection
def _assert_elements_are( self, collection, *args ):
elements = collection["elements"]
self.assertEquals(len(elements), len(args))
for index, element in enumerate(elements):
arg = args[index]
self.assertEquals(arg, element["element_identifier"])
return elements
def _verify_element( self, history_id, element, **props ):
object_id = element["object"]["id"]
if "contents" in props:
expected_contents = props["contents"]
contents = self.dataset_populator.get_history_dataset_content( history_id, dataset_id=object_id)
self.assertEquals( contents, expected_contents )
del props["contents"]
if props:
details = self.dataset_populator.get_history_dataset_details( history_id, dataset_id=object_id)
for key, value in props.items():
self.assertEquals( details[key], value )
def _setup_repeat_multirun( self ):
history_id = self.dataset_populator.new_history()
new_dataset1 = self.dataset_populator.new_dataset( history_id, content='123' )
@@ -9,7 +9,9 @@
<param name="input1" type="data_collection" collection_type="list" label="Input" help="Input collection..." format="txt" />
</inputs>
<outputs>
<collection name="list_output" type="list" label="Duplicate List" structured_like="input1" format="txt">
<collection name="list_output" type="list" label="Duplicate List" structured_like="input1" inherit_format="true">
<!-- inherit_format can be used in conjunction with structured_like
to perserve format. -->
</collection>
</outputs>
<tests>
@@ -0,0 +1,22 @@
<tool id="collection_creates_list_2" name="collection_creates_list_2" version="0.1.0">
<!-- go through and a header to each item in a collection - should use implicit
mapping the non-collectiony add header tool to do this in a real analysis.
-->
<command>
#for $key in $list_output.keys()#
cat "$header" > "$list_output[$key]";
cat "$input_collect[$key]" >> "$list_output[$key]";
#end for#
echo 'ensure not empty';
</command>
<inputs>
<param name="header" type="data" label="Input Data" help="Input data..." />
<param name="input_collect" type="data_collection" collection_type="list" label="Input Collect" help="Input collection..." />
</inputs>
<outputs>
<collection name="list_output" type="list" label="Duplicate List" structured_like="input_collect" format_source="header">
</collection>
</outputs>
<tests>
</tests>
</tool>
@@ -8,9 +8,13 @@
</inputs>
<outputs>
<collection name="paired_output" type="paired" label="Split Pair">
<!-- can reference parts directly or find via from_work_dir. -->
<!-- command can reference parts directly or find via from_work_dir. -->
<data name="forward" format="txt" />
<data name="reverse" format="txt" from_work_dir="reverse.txt" />
<data name="reverse" format_source="input1" from_work_dir="reverse.txt" />
<!-- data elements can use format, format_source, metadata_from,
from_work_dir. The format="input" idiom is not supported,
it should be considered deprecated and format_source is superior.
-->
</collection>
</outputs>
<tests>
@@ -36,6 +36,7 @@
<tool file="collection_two_paired.xml" />
<tool file="collection_creates_pair.xml" />
<tool file="collection_creates_list.xml" />
<tool file="collection_creates_list_2.xml" />
<tool file="collection_optional_param.xml" />
<tool file="collection_split_on_column.xml" />