From ddeff4ec0cdb0c6a803ae09843d0c70e2a70556b Mon Sep 17 00:00:00 2001 From: John Chilton Date: Sun, 13 Dec 2015 11:01:06 +0000 Subject: [PATCH] New model tool that filters failed datasets out of a collection. This differs from a traditional tool in that its inputs don't need to be in an 'ok' state and instead of creating new datasets and duplicating data on disk, new HDAs are created from the existing datasets. --- lib/galaxy/model/__init__.py | 4 ++ lib/galaxy/tools/__init__.py | 19 ++++++++ lib/galaxy/tools/filter_failed_collection.xml | 45 +++++++++++++++++++ lib/galaxy/tools/special_tools.py | 1 + test/api/helpers.py | 12 +++-- test/api/test_tools.py | 31 +++++++++++++ test/functional/tools/exit_code_from_file.xml | 13 ++++++ test/functional/tools/samples_tool_conf.xml | 1 + 8 files changed, 123 insertions(+), 3 deletions(-) create mode 100644 lib/galaxy/tools/filter_failed_collection.xml create mode 100644 test/functional/tools/exit_code_from_file.xml diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 352467b04d8..becda4b1455 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -2125,6 +2125,10 @@ class DatasetInstance( object ): return False return True + @property + def is_ok(self): + return self.state == self.states.OK + @property def is_pending( self ): """ diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index f8c57d9112e..a597b8cc4a1 100755 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -2235,6 +2235,25 @@ class ZipCollectionTool( DatabaseOperationTool ): ) +class FilterFailedDatasetsTool( DatabaseOperationTool ): + tool_type = 'filter_failed_datasets_collection' + require_dataset_ok = False + + def produce_outputs( self, trans, out_data, output_collections, incoming, history ): + hdca = incoming[ "input" ] + assert hdca.collection.collection_type == "list" + new_elements = odict() + for dce in hdca.collection.elements: + element = dce.element_object + if element.is_ok: + element_identifier = dce.element_identifier + new_elements[element_identifier] = element.copy() + + output_collections.create_collection( + self.outputs.values()[0], "output", elements=new_elements + ) + + # Populate tool_type to ToolClass mappings tool_types = {} for tool_class in [ Tool, SetMetadataTool, OutputParameterJSONTool, diff --git a/lib/galaxy/tools/filter_failed_collection.xml b/lib/galaxy/tools/filter_failed_collection.xml new file mode 100644 index 00000000000..7bd2e93dd2d --- /dev/null +++ b/lib/galaxy/tools/filter_failed_collection.xml @@ -0,0 +1,45 @@ + + datasets from a collection + + + + + + + + + + + + + + + + + + + + + + + + + + + + This tool takes a list dataset collction and filters out the failed + datasets from it. This is useful for continuing a multi-sample analysis + when one of more of the samples fails at some point. + + This tool will create new history datasets from your collection + but your quota usage will not increase. + + diff --git a/lib/galaxy/tools/special_tools.py b/lib/galaxy/tools/special_tools.py index b14415c8f29..4b5c8e7d897 100644 --- a/lib/galaxy/tools/special_tools.py +++ b/lib/galaxy/tools/special_tools.py @@ -6,6 +6,7 @@ SPECIAL_TOOLS = { "history import": "galaxy/tools/imp_exp/imp_history_from_archive.xml", "collection unzip": "galaxy/tools/unzip_collection.xml", "collection zip": "galaxy/tools/zip_collection.xml", + "filter failed datasets": "galaxy/tools/filter_failed_collection.xml", } diff --git a/test/api/helpers.py b/test/api/helpers.py index a72c5dc0051..02ea2717bce 100644 --- a/test/api/helpers.py +++ b/test/api/helpers.py @@ -145,18 +145,24 @@ class BaseDatasetPopulator( object ): return tool_response.json() def get_history_dataset_content( self, history_id, wait=True, **kwds ): - dataset_id = self.__history_dataset_id( history_id, wait=wait, **kwds ) + dataset_id = self.__history_content_id( history_id, wait=wait, **kwds ) display_response = self.__get_contents_request( history_id, "/%s/display" % dataset_id ) assert display_response.status_code == 200, display_response.content return display_response.content def get_history_dataset_details( self, history_id, **kwds ): - dataset_id = self.__history_dataset_id( history_id, **kwds ) + dataset_id = self.__history_content_id( history_id, **kwds ) details_response = self.__get_contents_request( history_id, "/datasets/%s" % dataset_id ) assert details_response.status_code == 200 return details_response.json() - def __history_dataset_id( self, history_id, wait=True, **kwds ): + def get_history_collection_details( self, history_id, **kwds ): + hdca_id = self.__history_content_id( history_id, **kwds ) + details_response = self.__get_contents_request( history_id, "/dataset_collections/%s" % hdca_id ) + assert details_response.status_code == 200, details_response.content + return details_response.json() + + def __history_content_id( self, history_id, wait=True, **kwds ): if wait: assert_ok = kwds.get( "assert_ok", True ) self.wait_for_history( history_id, assert_ok=assert_ok ) diff --git a/test/api/test_tools.py b/test/api/test_tools.py index 190fe6245f7..17b1761955a 100644 --- a/test/api/test_tools.py +++ b/test/api/test_tools.py @@ -165,6 +165,37 @@ class ToolsTestCase( api.ApiTestCase ): implicit_collections = response[ "implicit_collections" ] self.assertEquals( len(implicit_collections), 1 ) + def test_filter_failed( self ): + history_id = self.dataset_populator.new_history() + ok_hdca_id = self.dataset_collection_populator.create_list_in_history( history_id, contents=["0", "1", "0", "1"] ).json()["id"] + exit_code_inputs = { + "input": { 'batch': True, 'values': [ {"src": "hdca", "id": ok_hdca_id} ] }, + } + response = self._run( "exit_code_from_file", history_id, exit_code_inputs, assert_ok=False ).json() + self.dataset_populator.wait_for_history( history_id, assert_ok=False ) + + mixed_implicit_collections = response[ "implicit_collections" ] + self.assertEquals( len(mixed_implicit_collections), 1 ) + mixed_hdca_hid = mixed_implicit_collections[0]["hid"] + mixed_hdca = self.dataset_populator.get_history_collection_details(history_id, hid=mixed_hdca_hid, wait=False) + + def get_state(dce): + return dce["object"]["state"] + + mixed_states = map(get_state, mixed_hdca["elements"]) + assert mixed_states == [u"ok", u"error", u"ok", u"error"], mixed_states + inputs = { + "input": { "src": "hdca", "id": mixed_hdca["id"] }, + } + response = self._run( "__FILTER_FAILED_DATASETS__", history_id, inputs, assert_ok=False ).json() + self.dataset_populator.wait_for_history( history_id, assert_ok=False ) + filter_output_collections = response[ "output_collections" ] + self.assertEquals( len(filter_output_collections), 1 ) + filtered_hid = filter_output_collections[0]["hid"] + filtered_hdca = self.dataset_populator.get_history_collection_details(history_id, hid=filtered_hid, wait=False) + filtered_states = map(get_state, filtered_hdca["elements"]) + assert filtered_states == [u"ok", u"ok"], filtered_states + @skip_without_tool( "multi_select" ) def test_multi_select_as_list( self ): history_id = self.dataset_populator.new_history() diff --git a/test/functional/tools/exit_code_from_file.xml b/test/functional/tools/exit_code_from_file.xml new file mode 100644 index 00000000000..5d66d96aa3c --- /dev/null +++ b/test/functional/tools/exit_code_from_file.xml @@ -0,0 +1,13 @@ + + + sh -c "exit `cat $input`" + + + + + + + + + + diff --git a/test/functional/tools/samples_tool_conf.xml b/test/functional/tools/samples_tool_conf.xml index bbbeed7395f..f41aed6ad32 100644 --- a/test/functional/tools/samples_tool_conf.xml +++ b/test/functional/tools/samples_tool_conf.xml @@ -35,6 +35,7 @@ --> +