New model tool that filters failed datasets out of a collection.

This differs from a traditional tool in that its inputs don't need to be in an 'ok' state and instead of creating new datasets and duplicating data on disk, new HDAs are created from the existing datasets.
This commit is contained in:
John Chilton
2016-06-10 13:34:18 -04:00
parent 6ddb2c2a6a
commit ddeff4ec0c
8 changed files with 123 additions and 3 deletions
+4
View File
@@ -2125,6 +2125,10 @@ class DatasetInstance( object ):
return False
return True
@property
def is_ok(self):
return self.state == self.states.OK
@property
def is_pending( self ):
"""
+19
View File
@@ -2235,6 +2235,25 @@ class ZipCollectionTool( DatabaseOperationTool ):
)
class FilterFailedDatasetsTool( DatabaseOperationTool ):
tool_type = 'filter_failed_datasets_collection'
require_dataset_ok = False
def produce_outputs( self, trans, out_data, output_collections, incoming, history ):
hdca = incoming[ "input" ]
assert hdca.collection.collection_type == "list"
new_elements = odict()
for dce in hdca.collection.elements:
element = dce.element_object
if element.is_ok:
element_identifier = dce.element_identifier
new_elements[element_identifier] = element.copy()
output_collections.create_collection(
self.outputs.values()[0], "output", elements=new_elements
)
# Populate tool_type to ToolClass mappings
tool_types = {}
for tool_class in [ Tool, SetMetadataTool, OutputParameterJSONTool,
@@ -0,0 +1,45 @@
<tool id="__FILTER_FAILED_DATASETS__"
name="Filter failed"
version="1.0.0"
tool_type="filter_failed_datasets_collection">
<description>datasets from a collection</description>
<type class="FilterFailedDatasetsTool" module="galaxy.tools" />
<action module="galaxy.tools.actions.model_operations"
class="ModelOperationToolAction"/>
<inputs>
<param type="data_collection" collection_type="list" name="input" label="Input Collection" />
</inputs>
<outputs>
<collection name="output" format_source="input" type_source="input" label="${on_string} (filtered failed datasets)" >
</collection>
</outputs>
<tests>
<!-- Test framework has no way of creating a collection with
failed elements, so best we can do is verify identity on
an okay collection. API tests verify this tool works
though.
-->
<test>
<param name="input">
<collection type="list">
<element name="e1" value="simple_line.txt" />
</collection>
</param>
<output_collection name="output" type="list">
<element name="e1">
<assert_contents>
<has_text_matching expression="^This is a line of text.\n$" />
</assert_contents>
</element>
</output_collection>
</test>
</tests>
<help>
This tool takes a list dataset collction and filters out the failed
datasets from it. This is useful for continuing a multi-sample analysis
when one of more of the samples fails at some point.
This tool will create new history datasets from your collection
but your quota usage will not increase.
</help>
</tool>
+1
View File
@@ -6,6 +6,7 @@ SPECIAL_TOOLS = {
"history import": "galaxy/tools/imp_exp/imp_history_from_archive.xml",
"collection unzip": "galaxy/tools/unzip_collection.xml",
"collection zip": "galaxy/tools/zip_collection.xml",
"filter failed datasets": "galaxy/tools/filter_failed_collection.xml",
}
+9 -3
View File
@@ -145,18 +145,24 @@ class BaseDatasetPopulator( object ):
return tool_response.json()
def get_history_dataset_content( self, history_id, wait=True, **kwds ):
dataset_id = self.__history_dataset_id( history_id, wait=wait, **kwds )
dataset_id = self.__history_content_id( history_id, wait=wait, **kwds )
display_response = self.__get_contents_request( history_id, "/%s/display" % dataset_id )
assert display_response.status_code == 200, display_response.content
return display_response.content
def get_history_dataset_details( self, history_id, **kwds ):
dataset_id = self.__history_dataset_id( history_id, **kwds )
dataset_id = self.__history_content_id( history_id, **kwds )
details_response = self.__get_contents_request( history_id, "/datasets/%s" % dataset_id )
assert details_response.status_code == 200
return details_response.json()
def __history_dataset_id( self, history_id, wait=True, **kwds ):
def get_history_collection_details( self, history_id, **kwds ):
hdca_id = self.__history_content_id( history_id, **kwds )
details_response = self.__get_contents_request( history_id, "/dataset_collections/%s" % hdca_id )
assert details_response.status_code == 200, details_response.content
return details_response.json()
def __history_content_id( self, history_id, wait=True, **kwds ):
if wait:
assert_ok = kwds.get( "assert_ok", True )
self.wait_for_history( history_id, assert_ok=assert_ok )
+31
View File
@@ -165,6 +165,37 @@ class ToolsTestCase( api.ApiTestCase ):
implicit_collections = response[ "implicit_collections" ]
self.assertEquals( len(implicit_collections), 1 )
def test_filter_failed( self ):
history_id = self.dataset_populator.new_history()
ok_hdca_id = self.dataset_collection_populator.create_list_in_history( history_id, contents=["0", "1", "0", "1"] ).json()["id"]
exit_code_inputs = {
"input": { 'batch': True, 'values': [ {"src": "hdca", "id": ok_hdca_id} ] },
}
response = self._run( "exit_code_from_file", history_id, exit_code_inputs, assert_ok=False ).json()
self.dataset_populator.wait_for_history( history_id, assert_ok=False )
mixed_implicit_collections = response[ "implicit_collections" ]
self.assertEquals( len(mixed_implicit_collections), 1 )
mixed_hdca_hid = mixed_implicit_collections[0]["hid"]
mixed_hdca = self.dataset_populator.get_history_collection_details(history_id, hid=mixed_hdca_hid, wait=False)
def get_state(dce):
return dce["object"]["state"]
mixed_states = map(get_state, mixed_hdca["elements"])
assert mixed_states == [u"ok", u"error", u"ok", u"error"], mixed_states
inputs = {
"input": { "src": "hdca", "id": mixed_hdca["id"] },
}
response = self._run( "__FILTER_FAILED_DATASETS__", history_id, inputs, assert_ok=False ).json()
self.dataset_populator.wait_for_history( history_id, assert_ok=False )
filter_output_collections = response[ "output_collections" ]
self.assertEquals( len(filter_output_collections), 1 )
filtered_hid = filter_output_collections[0]["hid"]
filtered_hdca = self.dataset_populator.get_history_collection_details(history_id, hid=filtered_hid, wait=False)
filtered_states = map(get_state, filtered_hdca["elements"])
assert filtered_states == [u"ok", u"ok"], filtered_states
@skip_without_tool( "multi_select" )
def test_multi_select_as_list( self ):
history_id = self.dataset_populator.new_history()
@@ -0,0 +1,13 @@
<tool id="exit_code_from_file" name="exit_code_from_file">
<command detect_errors="exit_code">
sh -c "exit `cat $input`"
</command>
<inputs>
<param name="input" type="data" label="Exit code file" />
</inputs>
<outputs>
<data name="out_file1" />
</outputs>
<help>
</help>
</tool>
@@ -35,6 +35,7 @@
<tool file="maxseconds.xml" />
-->
<tool file="job_properties.xml" />
<tool file="exit_code_from_file.xml" />
<tool file="gzipped_inputs.xml" />
<tool file="output_order.xml" />
<tool file="output_format.xml" />