From 0ff639a29c6179e530c977b358cebf229c76139d Mon Sep 17 00:00:00 2001 From: John Chilton Date: Fri, 13 Oct 2017 20:43:24 -0400 Subject: [PATCH 01/22] More debugging in workflows API tests. --- test/api/test_workflows.py | 30 +++++++++++++++--------------- 1 file changed, 15 insertions(+), 15 deletions(-) diff --git a/test/api/test_workflows.py b/test/api/test_workflows.py index 4b8c481043e..ea539743e21 100644 --- a/test/api/test_workflows.py +++ b/test/api/test_workflows.py @@ -734,8 +734,8 @@ steps: def test_workflow_run_dynamic_output_collections_2(self): # A more advanced output collection workflow, testing regression of # https://github.com/galaxyproject/galaxy/issues/776 - history_id = self.dataset_populator.new_history() - workflow_id = self._upload_yaml_workflow(""" + with self.dataset_populator.test_history() as history_id: + workflow_id = self._upload_yaml_workflow(""" class: GalaxyWorkflow steps: - label: test_input_1 @@ -759,19 +759,19 @@ steps: - input2: $link: split_up#split_output """) - hda1 = self.dataset_populator.new_dataset(history_id, content="samp1\t10.0\nsamp2\t20.0\n") - hda2 = self.dataset_populator.new_dataset(history_id, content="samp1\t20.0\nsamp2\t40.0\n") - hda3 = self.dataset_populator.new_dataset(history_id, content="samp1\t30.0\nsamp2\t60.0\n") - self.dataset_populator.wait_for_history(history_id, assert_ok=True) - inputs = { - '0': self._ds_entry(hda1), - '1': self._ds_entry(hda2), - '2': self._ds_entry(hda3), - } - invocation_id = self.__invoke_workflow(history_id, workflow_id, inputs) - self.wait_for_invocation_and_jobs(history_id, workflow_id, invocation_id) - content = self.dataset_populator.get_history_dataset_content(history_id, hid=7) - self.assertEqual(content.strip(), "samp1\t10.0\nsamp2\t20.0") + hda1 = self.dataset_populator.new_dataset(history_id, content="samp1\t10.0\nsamp2\t20.0\n") + hda2 = self.dataset_populator.new_dataset(history_id, content="samp1\t20.0\nsamp2\t40.0\n") + hda3 = self.dataset_populator.new_dataset(history_id, content="samp1\t30.0\nsamp2\t60.0\n") + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + inputs = { + '0': self._ds_entry(hda1), + '1': self._ds_entry(hda2), + '2': self._ds_entry(hda3), + } + invocation_id = self.__invoke_workflow(history_id, workflow_id, inputs) + self.wait_for_invocation_and_jobs(history_id, workflow_id, invocation_id) + content = self.dataset_populator.get_history_dataset_content(history_id, hid=7) + self.assertEqual(content.strip(), "samp1\t10.0\nsamp2\t20.0") @skip_without_tool("collection_split_on_column") def test_workflow_run_dynamic_output_collections_3(self): From 4e9ee22ebcd5cc358d1b86f60e957846f35765a5 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Mon, 16 Oct 2017 18:58:37 -0400 Subject: [PATCH 02/22] More verbose workflow extraction test cases. --- test/api/test_workflow_extraction.py | 6 +++++- test/base/populators.py | 12 ++++++++++++ 2 files changed, 17 insertions(+), 1 deletion(-) diff --git a/test/api/test_workflow_extraction.py b/test/api/test_workflow_extraction.py index ac763f36df5..57008511576 100644 --- a/test/api/test_workflow_extraction.py +++ b/test/api/test_workflow_extraction.py @@ -5,7 +5,7 @@ import operator from collections import namedtuple from json import dumps, loads -from base.populators import skip_without_tool +from base.populators import skip_without_tool, summarize_instance_history_on_error from .test_workflows import BaseWorkflowsApiTestCase @@ -17,6 +17,7 @@ class WorkflowExtractionApiTestCase(BaseWorkflowsApiTestCase): self.history_id = self.dataset_populator.new_history() @skip_without_tool("cat1") + @summarize_instance_history_on_error def test_extract_from_history(self): # Run the simple test workflow and extract it back out from history cat1_job_id = self.__setup_and_run_cat1_workflow(history_id=self.history_id) @@ -29,6 +30,7 @@ class WorkflowExtractionApiTestCase(BaseWorkflowsApiTestCase): self.assertEqual(downloaded_workflow["name"], "test import from history") self.__assert_looks_like_cat1_example_workflow(downloaded_workflow) + @summarize_instance_history_on_error def test_extract_with_copied_inputs(self): old_history_id = self.dataset_populator.new_history() # Run the simple test workflow and extract it back out from history @@ -54,6 +56,7 @@ class WorkflowExtractionApiTestCase(BaseWorkflowsApiTestCase): self.__assert_looks_like_cat1_example_workflow(downloaded_workflow) @skip_without_tool("random_lines1") + @summarize_instance_history_on_error def test_extract_mapping_workflow_from_history(self): hdca, job_id1, job_id2 = self.__run_random_lines_mapped_over_pair(self.history_id) downloaded_workflow = self._extract_and_download_workflow( @@ -232,6 +235,7 @@ test_data: ) @skip_without_tool("collection_creates_pair") + @summarize_instance_history_on_error def test_extract_with_mapped_output_collections(self): jobs_summary = self._run_jobs(""" class: GalaxyWorkflow diff --git a/test/base/populators.py b/test/base/populators.py index c6e364217aa..2beaaf0b29f 100644 --- a/test/base/populators.py +++ b/test/base/populators.py @@ -92,6 +92,18 @@ def skip_without_datatype(extension): return method_wrapper +def summarize_instance_history_on_error(method): + @wraps(method) + def wrapped_method(api_test_case, *args, **kwds): + try: + method(api_test_case, *args, **kwds) + except Exception: + api_test_case.dataset_populator._summarize_history(api_test_case.history_id) + raise + + return wrapped_method + + def _raise_skip_if(check): if check: from nose.plugins.skip import SkipTest From 0e0894bbba8ee1d8b7c6946a5a42695fa6bd0858 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Wed, 18 Oct 2017 13:20:55 -0400 Subject: [PATCH 03/22] Refactoring history contents API. --- .../webapps/galaxy/api/history_contents.py | 52 +++++++++---------- 1 file changed, 25 insertions(+), 27 deletions(-) diff --git a/lib/galaxy/webapps/galaxy/api/history_contents.py b/lib/galaxy/webapps/galaxy/api/history_contents.py index f881fd8822f..87954ef4542 100644 --- a/lib/galaxy/webapps/galaxy/api/history_contents.py +++ b/lib/galaxy/webapps/galaxy/api/history_contents.py @@ -133,26 +133,33 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary @expose_api_anonymous def show(self, trans, id, history_id, **kwd): """ - show( self, trans, id, history_id, **kwd ) * GET /api/histories/{history_id}/contents/{id} - return detailed information about an HDA within a history + * GET /api/histories/{history_id}/contents/{type}/{id} + return detailed information about an HDA or HDCA within a history .. note:: Anonymous users are allowed to get their current history contents :type id: str - :param id: the encoded id of the HDA to return + :param id: the encoded id of the HDA or HDCA to return + :type type: str + :param id: 'dataset' or 'dataset_collection' :type history_id: str - :param history_id: encoded id string of the HDA's History + :param history_id: encoded id string of the HDA's or HDCA's History :rtype: dict - :returns: dictionary containing detailed HDA information + :returns: dictionary containing detailed HDA or HDCA information """ - contents_type = kwd.get('type', 'dataset') + contents_type = self.__get_contents_type(trans, kwd) if contents_type == 'dataset': return self.__show_dataset(trans, id, **kwd) elif contents_type == 'dataset_collection': return self.__show_dataset_collection(trans, id, history_id, **kwd) - else: - return self.__handle_unknown_contents_type(trans, contents_type) + + def __get_contents_type(self, trans, kwd): + contents_type = kwd.get('type', 'dataset') + if contents_type not in ['dataset', 'dataset_collection']: + self.__handle_unknown_contents_type(trans, contents_type) + + return contents_type def __show_dataset(self, trans, id, **kwd): hda = self.hda_manager.get_accessible(self.decode_id(id), trans.user) @@ -162,18 +169,15 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary **self._parse_serialization_params(kwd, 'detailed')) def __show_dataset_collection(self, trans, id, history_id, **kwd): - try: - service = trans.app.dataset_collections_service - dataset_collection_instance = service.get_dataset_collection_instance( - trans=trans, - instance_type='history', - id=id, - ) - return self.__collection_dict(trans, dataset_collection_instance, view="element") - except Exception as e: - log.exception("Error in history API at listing dataset collection") - trans.response.status = 500 - return {'error': str(e)} + dataset_collection_instance = self.__get_accessible_collection(trans, id, history_id) + return self.__collection_dict(trans, dataset_collection_instance, view="element") + + def __get_accessible_collection(self, trans, id, history_id): + return trans.app.dataset_collections_service.get_dataset_collection_instance( + trans=trans, + instance_type="history", + id=id + ) @expose_api_raw_anonymous def download_dataset_collection(self, trans, id, history_id=None, **kwd): @@ -188,14 +192,8 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary :param history_id: encoded id string of the HDCA's History """ try: - service = trans.app.dataset_collections_service - dataset_collection_instance = service.get_dataset_collection_instance( - trans=trans, - instance_type='history', - id=id, - ) + dataset_collection_instance = self.__get_accessible_collection(trans, id, history_id) return self.__stream_dataset_collection(trans, dataset_collection_instance) - except Exception as e: log.exception("Error in API while creating dataset collection archive") trans.response.status = 500 From 8c1658f61a12b9c1e2efdb04ee3530c1272b0bde Mon Sep 17 00:00:00 2001 From: John Chilton Date: Mon, 16 Oct 2017 22:14:06 -0400 Subject: [PATCH 04/22] Mixin for uses create and update time. --- lib/galaxy/jobs/__init__.py | 2 +- lib/galaxy/model/__init__.py | 26 +++++++++++++++----------- 2 files changed, 16 insertions(+), 12 deletions(-) diff --git a/lib/galaxy/jobs/__init__.py b/lib/galaxy/jobs/__init__.py index 8c82a17da06..3afd5657a3a 100644 --- a/lib/galaxy/jobs/__init__.py +++ b/lib/galaxy/jobs/__init__.py @@ -1040,7 +1040,7 @@ class JobWrapper(object, HasResourceParameters): destination_params = job.destination_params if "__resubmit_delay_seconds" in destination_params: delay = float(destination_params["__resubmit_delay_seconds"]) - if job.seconds_since_update < delay: + if job.seconds_since_updated < delay: return False return True diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 01689a1a555..f054edac4e3 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -117,6 +117,19 @@ class HasName: return name +class UsesCreateAndUpdateTime: + + @property + def seconds_since_updated(self): + update_time = self.update_time or galaxy.model.orm.now.now() # In case not yet flushed + return (galaxy.model.orm.now.now() - update_time).total_seconds() + + @property + def seconds_since_created(self): + create_time = self.create_time or galaxy.model.orm.now.now() # In case not yet flushed + return (galaxy.model.orm.now.now() - create_time).total_seconds() + + class JobLike: def _init_metrics(self): @@ -424,7 +437,7 @@ class TaskMetricNumeric(BaseJobMetric): pass -class Job(object, JobLike, Dictifiable): +class Job(object, JobLike, UsesCreateAndUpdateTime, Dictifiable): dict_collection_visible_keys = ['id', 'state', 'exit_code', 'update_time', 'create_time'] dict_element_visible_keys = ['id', 'state', 'exit_code', 'update_time', 'create_time'] @@ -805,10 +818,6 @@ class Job(object, JobLike, Dictifiable): config_value = default return config_value - @property - def seconds_since_update(self): - return (galaxy.model.orm.now.now() - self.update_time).total_seconds() - class Task(object, JobLike): """ @@ -3952,7 +3961,7 @@ class StoredWorkflowMenuEntry(object): self.order_index = None -class WorkflowInvocation(object, Dictifiable): +class WorkflowInvocation(object, UsesCreateAndUpdateTime, Dictifiable): dict_collection_visible_keys = ['id', 'update_time', 'workflow_id', 'history_id', 'uuid', 'state'] dict_element_visible_keys = ['id', 'update_time', 'workflow_id', 'history_id', 'uuid', 'state'] states = Bunch( @@ -4121,11 +4130,6 @@ class WorkflowInvocation(object, Dictifiable): return True return False - @property - def seconds_since_created(self): - create_time = self.create_time or galaxy.model.orm.now.now() # In case not flushed yet - return (galaxy.model.orm.now.now() - create_time).total_seconds() - class WorkflowInvocationToSubworkflowInvocationAssociation(object, Dictifiable): dict_collection_visible_keys = ['id', 'workflow_step_id', 'workflow_invocation_id', 'subworkflow_invocation_id'] From c076b3b6c5bf5ac9d374d56d42c9a91c352688ce Mon Sep 17 00:00:00 2001 From: John Chilton Date: Tue, 17 Oct 2017 08:41:18 -0400 Subject: [PATCH 05/22] Refactor workflow testing toward reuse via "populators". Should allow more functionality to be shared between vanilla API tests and integration tests for various workflow scheduling options. --- test/api/test_workflows.py | 15 ++------------- test/base/populators.py | 16 +++++++++++++++- 2 files changed, 17 insertions(+), 14 deletions(-) diff --git a/test/api/test_workflows.py b/test/api/test_workflows.py index ea539743e21..f875c2c9fa9 100644 --- a/test/api/test_workflows.py +++ b/test/api/test_workflows.py @@ -2040,19 +2040,8 @@ steps: self._assert_status_code_is(hda_info_response, 200) self.assertEqual(hda_info_response.json()["metadata_data_lines"], lines) - def __invoke_workflow(self, history_id, workflow_id, inputs={}, request={}, assert_ok=True): - request["history"] = "hist_id=%s" % history_id, - if inputs: - request["inputs"] = dumps(inputs) - request["inputs_by"] = 'step_index' - url = "workflows/%s/usage" % (workflow_id) - invocation_response = self._post(url, data=request) - if assert_ok: - self._assert_status_code_is(invocation_response, 200) - invocation_id = invocation_response.json()["id"] - return invocation_id - else: - return invocation_response + def __invoke_workflow(self, *args, **kwds): + return self.workflow_populator.invoke_workflow(*args, **kwds) def __import_workflow(self, workflow_id, deprecated_route=False): if deprecated_route: diff --git a/test/base/populators.py b/test/base/populators.py index 2beaaf0b29f..e382ace1ead 100644 --- a/test/base/populators.py +++ b/test/base/populators.py @@ -383,7 +383,21 @@ class BaseWorkflowPopulator(object): """ Wait for a workflow invocation to completely schedule and then history to be complete. """ self.wait_for_invocation(workflow_id, invocation_id, timeout=timeout) - self.dataset_populator.wait_for_history(history_id, assert_ok=assert_ok, timeout=timeout) + self.dataset_populator.wait_for_history_jobs(history_id, assert_ok=assert_ok, timeout=timeout) + + def invoke_workflow(self, history_id, workflow_id, inputs={}, request={}, assert_ok=True): + request["history"] = "hist_id=%s" % history_id, + if inputs: + request["inputs"] = json.dumps(inputs) + request["inputs_by"] = 'step_index' + url = "workflows/%s/usage" % (workflow_id) + invocation_response = self._post(url, data=request) + if assert_ok: + api_asserts.assert_status_code_is(invocation_response, 200) + invocation_id = invocation_response.json()["id"] + return invocation_id + else: + return invocation_response class WorkflowPopulator(BaseWorkflowPopulator, ImporterGalaxyInterface): From 55aa7bce5540469fad266f1166c9a2e7365030d4 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Mon, 16 Oct 2017 21:11:11 -0400 Subject: [PATCH 06/22] Test cases for mapping for outputs with filter statements. Checked through tools-iuc and I couldn't find any tools that use the actual input data parameters in filter statements so I think it is a fairly good approximation to assume that all mapped jobs will either filter or not at least with current execution model. --- test/api/test_tools.py | 38 +++++++++++++++++++ .../tools/output_filter_with_input.xml | 29 ++++++++++++++ test/functional/tools/samples_tool_conf.xml | 1 + 3 files changed, 68 insertions(+) create mode 100644 test/functional/tools/output_filter_with_input.xml diff --git a/test/api/test_tools.py b/test/api/test_tools.py index 306bfdca96f..e11ce93c8be 100644 --- a/test/api/test_tools.py +++ b/test/api/test_tools.py @@ -704,6 +704,38 @@ class ToolsTestCase(api.ApiTestCase): assert output1_details["file_ext"] == "txt" if (use_action == "do") else "data" assert output2_details["file_ext"] == "txt" if (use_action == "do") else "data" + @skip_without_tool("output_filter_with_input") + def test_map_over_with_output_filter_no_filtering(self): + with self.dataset_populator.test_history() as history_id: + hdca_id = self.dataset_collection_populator.create_list_in_history(history_id).json()["id"] + inputs = { + "input_1": {'batch': True, 'values': [{'src': 'hdca', 'id': hdca_id}]}, + "produce_out_1": "true", + "filter_text_1": "foo", + } + create = self._run('output_filter_with_input', history_id, inputs).json() + jobs = create['jobs'] + implicit_collections = create['implicit_collections'] + self.assertEquals(len(jobs), 3) + self.assertEquals(len(implicit_collections), 3) + self._check_implicit_collection_populated(create) + + @skip_without_tool("output_filter_with_input") + def test_map_over_with_output_filter_one_filtered(self): + with self.dataset_populator.test_history() as history_id: + hdca_id = self.dataset_collection_populator.create_list_in_history(history_id).json()["id"] + inputs = { + "input_1": {'batch': True, 'values': [{'src': 'hdca', 'id': hdca_id}]}, + "produce_out_1": "true", + "filter_text_1": "bar", + } + create = self._run('output_filter_with_input', history_id, inputs).json() + jobs = create['jobs'] + implicit_collections = create['implicit_collections'] + self.assertEquals(len(jobs), 3) + self.assertEquals(len(implicit_collections), 2) + self._check_implicit_collection_populated(create) + @skip_without_tool("Cut1") def test_map_over_with_complex_output_actions(self): history_id = self.dataset_populator.new_history() @@ -1297,6 +1329,12 @@ class ToolsTestCase(api.ApiTestCase): assert output1_content.strip() == "123\n456\nxxx", output1_content assert output2_content.strip() == "789\n0ab\nyyy", output2_content + def _check_implicit_collection_populated(self, run_response): + implicit_collections = run_response["implicit_collections"] + assert implicit_collections + for implicit_collection in implicit_collections: + assert implicit_collection["populated_state"] == "ok" + def _cat1_outputs(self, history_id, inputs): return self._run_outputs(self._run_cat1(history_id, inputs)) diff --git a/test/functional/tools/output_filter_with_input.xml b/test/functional/tools/output_filter_with_input.xml new file mode 100644 index 00000000000..9eb7dbc20d1 --- /dev/null +++ b/test/functional/tools/output_filter_with_input.xml @@ -0,0 +1,29 @@ + + + + echo "test" > 1; + echo "test" > 2; + echo "test" > 3; + echo "test" > 4; + echo "test" > 5; + + + + + + + + + produce_out_1 is True + + + filter_text_1 in ["foo", "bar"] + + filter_text_1 == "foo" + + + + + + + diff --git a/test/functional/tools/samples_tool_conf.xml b/test/functional/tools/samples_tool_conf.xml index fa5c8ed5cd6..88ecc25f816 100644 --- a/test/functional/tools/samples_tool_conf.xml +++ b/test/functional/tools/samples_tool_conf.xml @@ -60,6 +60,7 @@ + From d5f98f49045bca351e982f17cac9386308089eeb Mon Sep 17 00:00:00 2001 From: John Chilton Date: Tue, 25 Jul 2017 12:41:54 +0200 Subject: [PATCH 07/22] Formalize workflow invocation and invocation step outputs. Workflow Invocations -------------------- The workflow invocation outputs half of this is relatively straight forward. It is modelled somewhat on job outputs, output datasets and output dataset collections are now tracked for each workflow invocation and exposed via the workflow invocation API. This required adding new tables (linked to WorkflowInvocations and WorkflowOutputs) that track these output associations. Previously one could imagine backtracking this information for simple tool steps via the WorkflowInvocationStep -> Job table, but for steps that have many jobs (i.e. mapping over a collection) or for non-tool steps such information was more difficult to recover (and simply couldn't be recovered from the API at all or even internally without significant knowledge of the underlying workflow). Workflow Invocation Steps ------------------------- Tracking the outputs of WorkflowInvocationSteps was not previously done at all, one would have to follow the Job table as well. A signficant downside to this is that one cannot map over empty collections in a workflow - since no such job would exist. Tracking job outputs for WorkflowInvocationSteps is not a simple matter of just attaching outputs to an existing table because we had no concept of a workflow step tracked - since there could be many WorklfowInvocationSteps corresponding to the same combination of WorkflowInvocation and WorkflowStep. That should feel wrong and that is because it is - when collections were added the possiblity of having many jobs for the same combination of WorkflowInvocation and WorkflowStep was added. I should have split WorkflowInvocationSteps into WorkflowInvocationSteps and WorkflowInvocationStepJobAssociations at that time but didn't. This commit now does it - effectively normalizing the ``workflow_invocation_step`` table by introducing the new ``workflow_invocation_step_job_association`` table. Splitting up the WorkflowInvocationStep table this way allows recovering the mapped over output (e.g. the implicitly created collection from all the jobs) as well the outputs from the individual jobs (by walking WorkflowInvocationStep -> WorkflowInvocationStepJobAssociation -> Job -> JobToOutput*Association). This split up involves failrly substantial changes to the workflow module interface. Any place a list of WorkflowInvocationSteps was assumed, I reworked it to just expect a single WorkflowInvocationStep. I vastly simplified recover_mapping to just use the persisted outputs (this was needed in order to also implment empty collection mapping in workflows). This also fixes a bug (or implements a missing feature) where Subworkflow moudles had no recover_mapping methods - so for instance if a tool that produces dynamic collections appeared anywhere in a workflow after a subworkflow step - that workflow would not complete scheduling properly. Now that we have a way to reference the set of jobs corresponding to a workflow step within an invocation, we can start to track partial scheduling of such steps. This is outlined in https://github.com/galaxyproject/galaxy/issues/3883 and refactoring toward this goal is included here - including adding a state to WorkflowInvocationStep so Galaxy can determine if it has started scheduling this step and an index when scheduling jobs so it can tell how far into a scheduling things have gone as well as augmenting the tool executor to take a maximum number of jobs to execute and allow recovery of existing jobs for collection building purposes. *Applications* These changes will enable: - A simple, consistent API for finding workflow outputs that can be consumed by Planemo for testing workflows. - Mapping over empty collections in workflows. - Re-scheduling workflow invocations that include subworkflow steps. - Partial scheduling within steps requiring a large number of jobs when scheduling workflow invocations. --- lib/galaxy/jobs/actions/post.py | 12 +- lib/galaxy/model/__init__.py | 172 ++++++++++++---- lib/galaxy/model/mapping.py | 97 ++++++++- .../versions/0136_record_workflow_outputs.py | 185 ++++++++++++++++++ lib/galaxy/tools/execute.py | 58 +++++- lib/galaxy/workflow/modules.py | 124 +++++++----- lib/galaxy/workflow/run.py | 108 +++++++--- test/api/test_workflows.py | 71 +++++++ test/base/populators.py | 2 + test/unit/workflows/test_workflow_progress.py | 65 +++--- 10 files changed, 738 insertions(+), 156 deletions(-) create mode 100644 lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py diff --git a/lib/galaxy/jobs/actions/post.py b/lib/galaxy/jobs/actions/post.py index 267a765e66f..138d4b8005b 100644 --- a/lib/galaxy/jobs/actions/post.py +++ b/lib/galaxy/jobs/actions/post.py @@ -269,10 +269,10 @@ class DeleteIntermediatesAction(DefaultJobAction): # concurrently, sometimes non-terminal steps won't be cleaned up # because of the lag in job state updates. sa_session.flush() - if not job.workflow_invocation_step: + if not job.workflow_invocation_step_assoc.workflow_invocation_step: log.debug("This job is not part of a workflow invocation, delete intermediates aborted.") return - wfi = job.workflow_invocation_step.workflow_invocation + wfi = job.workflow_invocation_step_assoc.workflow_invocation_step.workflow_invocation sa_session.refresh(wfi) if wfi.active: log.debug("Workflow still scheduling so new jobs may appear, skipping deletion of intermediate files.") @@ -285,9 +285,9 @@ class DeleteIntermediatesAction(DefaultJobAction): jobs_to_check = [] for wfi_step in wfi_steps: sa_session.refresh(wfi_step) - wfi_step_job = wfi_step.job - if wfi_step_job: - jobs_to_check.append(wfi_step_job) + wfi_step_job_assocs = wfi_step.jobs + if wfi_step_job_assocs: + jobs_to_check.extend(map(lambda j: j.job, wfi_step_job_assocs)) else: log.debug("No job found yet for wfi_step %s, (step %s)" % (wfi_step, wfi_step.workflow_step)) for j2c in jobs_to_check: @@ -302,7 +302,7 @@ class DeleteIntermediatesAction(DefaultJobAction): for (input_dataset, creating_job) in creating_jobs: sa_session.refresh(creating_job) sa_session.refresh(input_dataset) - for input_dataset in [x.dataset for (x, creating_job) in creating_jobs if creating_job.workflow_invocation_step and creating_job.workflow_invocation_step.workflow_invocation == wfi]: + for input_dataset in [x.dataset for (x, creating_job) in creating_jobs if creating_job.workflow_invocation_step_assoc and creating_job.workflow_invocation_step_assoc.workflow_invocation_step.workflow_invocation == wfi]: # note that the above input_dataset is a reference to a # job.input_dataset.dataset at this point safe_to_delete = True diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index f054edac4e3..707f96cbdbf 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -803,8 +803,9 @@ class Job(object, JobLike, UsesCreateAndUpdateTime, Dictifiable): def set_final_state(self, final_state): self.set_state(final_state) - if self.workflow_invocation_step: - self.workflow_invocation_step.update() + workflow_invocation_step_assoc = self.workflow_invocation_step_assoc + if workflow_invocation_step_assoc: + workflow_invocation_step_assoc.workflow_invocation_step.update() def get_destination_configuration(self, config, key, default=None): """ Get a destination parameter that can be defaulted back @@ -4036,17 +4037,16 @@ class WorkflowInvocation(object, UsesCreateAndUpdateTime, Dictifiable): step_invocations = {} for invocation_step in self.steps: step_id = invocation_step.workflow_step_id - if step_id not in step_invocations: - step_invocations[step_id] = [] - step_invocations[step_id].append(invocation_step) + assert step_id not in step_invocations + step_invocations[step_id] = invocation_step return step_invocations - def step_invocations_for_step_id(self, step_id): - step_invocations = [] + def step_invocation_for_step_id(self, step_id): + target_invocation_step = None for invocation_step in self.steps: if step_id == invocation_step.workflow_step_id: - step_invocations.append(invocation_step) - return step_invocations + target_invocation_step = invocation_step + return target_invocation_step @staticmethod def poll_active_workflow_ids( @@ -4072,6 +4072,24 @@ class WorkflowInvocation(object, UsesCreateAndUpdateTime, Dictifiable): # is relatively intutitive. return [wid for wid in query.all()] + def add_output(self, workflow_output, step, output_object): + if output_object.history_content_type == "dataset": + output_assoc = WorkflowInvocationOutputDatasetAssociation() + output_assoc.workflow_invocation = self + output_assoc.workflow_output = workflow_output + output_assoc.workflow_step = step + output_assoc.dataset = output_object + self.output_datasets.append(output_assoc) + elif output_object.history_content_type == "dataset_collection": + output_assoc = WorkflowInvocationOutputDatasetCollectionAssociation() + output_assoc.workflow_invocation = self + output_assoc.workflow_output = workflow_output + output_assoc.workflow_step = step + output_assoc.dataset_collection = output_object + self.output_dataset_collections.append(output_assoc) + else: + raise Exception("Uknown output type encountered") + def to_dict(self, view='collection', value_mapper=None, step_details=False): rval = super(WorkflowInvocation, self).to_dict(view=view, value_mapper=value_mapper) if view == 'element': @@ -4087,17 +4105,43 @@ class WorkflowInvocation(object, UsesCreateAndUpdateTime, Dictifiable): inputs = {} for step in self.steps: if step.workflow_step.type == 'tool': - for step_input in step.workflow_step.input_connections: - output_step_type = step_input.output_step.type - if output_step_type in ['data_input', 'data_collection_input']: - src = "hda" if output_step_type == 'data_input' else 'hdca' - for job_input in step.job.input_datasets: - if job_input.name == step_input.input_name: - inputs[str(step_input.output_step.order_index)] = { - "id": job_input.dataset_id, "src": src, - "uuid" : str(job_input.dataset.dataset.uuid) if job_input.dataset.dataset.uuid is not None else None - } + for step_job_assoc in step.jobs: + for step_input in step.workflow_step.input_connections: + output_step_type = step_input.output_step.type + if output_step_type in ['data_input', 'data_collection_input']: + src = "hda" if output_step_type == 'data_input' else 'hdca' + for job_input in step_job_assoc.job.input_datasets: + if job_input.name == step_input.input_name: + inputs[str(step_input.output_step.order_index)] = { + "id": job_input.dataset_id, "src": src, + "uuid" : str(job_input.dataset.dataset.uuid) if job_input.dataset.dataset.uuid is not None else None + } rval['inputs'] = inputs + + outputs = {} + for output_assoc in self.output_datasets: + label = output_assoc.workflow_output.label + if not label: + continue + + outputs[label] = { + 'src': 'hda', + 'id': output_assoc.dataset_id, + } + + output_collections = {} + for output_assoc in self.output_dataset_collections: + label = output_assoc.workflow_output.label + if not label: + continue + + output_collections[label] = { + 'src': 'hdca', + 'id': output_assoc.dataset_collection_id, + } + + rval['outputs'] = outputs + rval['output_collections'] = output_collections return rval def update(self): @@ -4137,36 +4181,70 @@ class WorkflowInvocationToSubworkflowInvocationAssociation(object, Dictifiable): class WorkflowInvocationStep(object, Dictifiable): - dict_collection_visible_keys = ['id', 'update_time', 'job_id', 'workflow_step_id', 'action'] - dict_element_visible_keys = ['id', 'update_time', 'job_id', 'workflow_step_id', 'action'] + dict_collection_visible_keys = ['id', 'update_time', 'job_id', 'workflow_step_id', 'state', 'action'] + dict_element_visible_keys = ['id', 'update_time', 'job_id', 'workflow_step_id', 'state', 'action'] + states = Bunch( + NEW='new', # Brand new workflow invocation step + READY='ready', # Workflow invocation step ready for another iteration of scheduling. + SCHEDULED='scheduled', # Workflow invocation step has been scheduled. + # CANCELLED='cancelled', TODO: implement and expose + # FAILED='failed', TODO: implement and expose + ) def update(self): self.workflow_invocation.update() + def add_output(self, output_name, output_object): + if output_object.history_content_type == "dataset": + output_assoc = WorkflowInvocationStepOutputDatasetAssociation() + output_assoc.workflow_invocation_step = self + output_assoc.dataset = output_object + output_assoc.output_name = output_name + self.output_datasets.append(output_assoc) + elif output_object.history_content_type == "dataset_collection": + output_assoc = WorkflowInvocationStepOutputDatasetCollectionAssociation() + output_assoc.workflow_invocation_step = self + output_assoc.dataset_collection = output_object + output_assoc.output_name = output_name + self.output_dataset_collections.append(output_assoc) + else: + raise Exception("Uknown output type encountered") + def to_dict(self, view='collection', value_mapper=None): rval = super(WorkflowInvocationStep, self).to_dict(view=view, value_mapper=value_mapper) rval['order_index'] = self.workflow_step.order_index rval['workflow_step_label'] = self.workflow_step.label rval['workflow_step_uuid'] = str(self.workflow_step.uuid) - rval['state'] = self.job.state if self.job is not None else None - if self.job is not None and view == 'element': - output_dict = {} - for i in self.job.output_datasets: - if i.dataset is not None: - output_dict[i.name] = { - "id" : i.dataset.id, "src" : "hda", - "uuid" : str(i.dataset.dataset.uuid) if i.dataset.dataset.uuid is not None else None - } - for i in self.job.output_library_datasets: - if i.dataset is not None: - output_dict[i.name] = { - "id" : i.dataset.id, "src" : "ldda", - "uuid" : str(i.dataset.dataset.uuid) if i.dataset.dataset.uuid is not None else None - } - rval['outputs'] = output_dict + # Following no longer makes sense... + # rval['state'] = self.job.state if self.job is not None else None + if view == 'element': + outputs = {} + for output_assoc in self.output_datasets: + name = output_assoc.output_name + outputs[name] = { + 'src': 'hda', + 'id': output_assoc.dataset.id, + 'uuid': str(output_assoc.dataset.dataset.uuid) if output_assoc.dataset.dataset.uuid is not None else None + } + + output_collections = {} + for output_assoc in self.output_dataset_collections: + name = output_assoc.output_name + output_collections[name] = { + 'src': 'hdca', + 'id': output_assoc.dataset_collection.id, + } + + rval['outputs'] = outputs + rval['output_collections'] = output_collections return rval +class WorkflowInvocationStepJobAssociation(object, Dictifiable): + dict_collection_visible_keys = ('id', 'job_id', 'workflow_invocation_step_id') + dict_element_visible_keys = ('id', 'job_id', 'workflow_invocation_step_id') + + class WorkflowRequest(object, Dictifiable): dict_collection_visible_keys = ['id', 'name', 'type', 'state', 'history_id', 'workflow_id'] dict_element_visible_keys = ['id', 'name', 'type', 'state', 'history_id', 'workflow_id'] @@ -4221,6 +4299,26 @@ class WorkflowRequestInputStepParmeter(object, Dictifiable): dict_collection_visible_keys = ['id', 'workflow_invocation_id', 'workflow_step_id', 'parameter_value'] +class WorkflowInvocationOutputDatasetAssociation(object, Dictifiable): + """Represents links to output datasets for the workflow.""" + dict_collection_visible_keys = ['id', 'workflow_invocation_id', 'workflow_step_id', 'dataset_id', 'name'] + + +class WorkflowInvocationOutputDatasetCollectionAssociation(object, Dictifiable): + """Represents links to output dataset collections for the workflow.""" + dict_collection_visible_keys = ['id', 'workflow_invocation_id', 'workflow_step_id', 'dataset_collection_id', 'name'] + + +class WorkflowInvocationStepOutputDatasetAssociation(object, Dictifiable): + """Represents links to output datasets for the workflow.""" + dict_collection_visible_keys = ['id', 'workflow_invocation_step_id', 'dataset_id', 'output_name'] + + +class WorkflowInvocationStepOutputDatasetCollectionAssociation(object, Dictifiable): + """Represents links to output dataset collections for the workflow.""" + dict_collection_visible_keys = ['id', 'workflow_invocation_step_id', 'dataset_collection_id', 'output_name'] + + class MetadataFile(StorableObject): def __init__(self, dataset=None, name=None): diff --git a/lib/galaxy/model/mapping.py b/lib/galaxy/model/mapping.py index 97923d892cc..49d6d4ed206 100644 --- a/lib/galaxy/model/mapping.py +++ b/lib/galaxy/model/mapping.py @@ -901,9 +901,54 @@ model.WorkflowInvocationStep.table = Table( Column("update_time", DateTime, default=now, onupdate=now), Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True, nullable=False), Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True, nullable=False), - Column("job_id", Integer, ForeignKey("job.id"), index=True, nullable=True), + Column("state", TrimmedString(64), index=True), Column("action", JSONType, nullable=True)) + +model.WorkflowInvocationStepJobAssociation.table = Table( + "workflow_invocation_step_job_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True, nullable=False), + Column("order_index", Integer, nullable=True), # recovering partially complete WorkflowInvocationSteps requires knowing which jobs have been scheduled + Column("job_id", Integer, ForeignKey("job.id"), index=True, nullable=False), +) + + +model.WorkflowInvocationOutputDatasetAssociation.table = Table( + "workflow_invocation_output_dataset_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True), + Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True), + Column("dataset_id", Integer, ForeignKey("history_dataset_association.id"), index=True), + Column("workflow_output_id", Integer, ForeignKey("workflow_output.id"), index=True), +) + +model.WorkflowInvocationOutputDatasetCollectionAssociation.table = Table( + "workflow_invocation_output_dataset_collection_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True), + Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True), + Column("dataset_collection_id", Integer, ForeignKey("history_dataset_collection_association.id"), index=True), + Column("workflow_output_id", Integer, ForeignKey("workflow_output.id"), index=True), +) + +model.WorkflowInvocationStepOutputDatasetAssociation.table = Table( + "workflow_invocation_step_output_dataset_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True), + Column("dataset_id", Integer, ForeignKey("history_dataset_association.id"), index=True), + Column("output_name", String(255), nullable=True), +) + +model.WorkflowInvocationStepOutputDatasetCollectionAssociation.table = Table( + "workflow_invocation_step_output_dataset_collection_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True), + Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True), + Column("dataset_collection_id", Integer, ForeignKey("history_dataset_collection_association.id"), index=True), + Column("output_name", String(255), nullable=True), +) + model.WorkflowInvocationToSubworkflowInvocationAssociation.table = Table( "workflow_invocation_to_subworkflow_invocation_association", metadata, Column("id", Integer, primary_key=True), @@ -2310,7 +2355,7 @@ mapper(model.WorkflowInvocation, model.WorkflowInvocation.table, properties=dict uselist=True, ), steps=relation(model.WorkflowInvocationStep, - backref='workflow_invocation'), + backref="workflow_invocation"), workflow=relation(model.Workflow) )) @@ -2323,12 +2368,17 @@ mapper(model.WorkflowInvocationToSubworkflowInvocationAssociation, model.Workflo workflow_step=relation(model.WorkflowStep), )) -mapper(model.WorkflowInvocationStep, model.WorkflowInvocationStep.table, properties=dict( - workflow_step=relation(model.WorkflowStep), + +simple_mapping(model.WorkflowInvocationStepJobAssociation, + workflow_invocation_step=relation(model.WorkflowInvocationStep, backref="jobs"), job=relation(model.Job, - backref=backref('workflow_invocation_step', - uselist=False)) -)) + backref=backref('workflow_invocation_step_assoc', + uselist=False))) + + +simple_mapping(model.WorkflowInvocationStep, + workflow_step=relation(model.WorkflowStep)) + simple_mapping(model.WorkflowRequestInputParameter, workflow_invocation=relation(model.WorkflowInvocation)) @@ -2358,6 +2408,39 @@ mapper(model.MetadataFile, model.MetadataFile.table, properties=dict( library_dataset=relation(model.LibraryDatasetDatasetAssociation) )) + +simple_mapping( + model.WorkflowInvocationOutputDatasetAssociation, + workflow_invocation=relation(model.WorkflowInvocation, backref="output_datasets"), + workflow_step=relation(model.WorkflowStep), + dataset=relation(model.HistoryDatasetAssociation), + workflow_output=relation(model.WorkflowOutput), +) + + +simple_mapping( + model.WorkflowInvocationOutputDatasetCollectionAssociation, + workflow_invocation=relation(model.WorkflowInvocation, backref="output_dataset_collections"), + workflow_step=relation(model.WorkflowStep), + dataset_collection=relation(model.HistoryDatasetCollectionAssociation), + workflow_output=relation(model.WorkflowOutput), +) + + +simple_mapping( + model.WorkflowInvocationStepOutputDatasetAssociation, + workflow_invocation_step=relation(model.WorkflowInvocationStep, backref="output_datasets"), + dataset=relation(model.HistoryDatasetAssociation), +) + + +simple_mapping( + model.WorkflowInvocationStepOutputDatasetCollectionAssociation, + workflow_invocation_step=relation(model.WorkflowInvocationStep, backref="output_dataset_collections"), + dataset_collection=relation(model.HistoryDatasetCollectionAssociation), +) + + mapper(model.PageRevision, model.PageRevision.table) mapper(model.Page, model.Page.table, properties=dict( diff --git a/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py b/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py new file mode 100644 index 00000000000..1b83a49aa31 --- /dev/null +++ b/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py @@ -0,0 +1,185 @@ +""" +Migration script for workflow request tables. +""" +from __future__ import print_function + +import datetime +import logging + +from collections import OrderedDict + +from migrate.changeset.constraint import ForeignKeyConstraint +from sqlalchemy import Column, DateTime, ForeignKey, Integer, MetaData, String, Table + +from galaxy.model.custom_types import JSONType, TrimmedString + + +now = datetime.datetime.utcnow + +log = logging.getLogger(__name__) +metadata = MetaData() + + +def get_new_tables(): + # Normally we define this globally in the file, but we need to delay the + # reading of existing tables because an existing workflow_invocation_step + # table exists that we want to recreate. + + workflow_invocation_output_dataset_association_table = Table( + "workflow_invocation_output_dataset_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True), + Column("workflow_step_id", Integer, ForeignKey("workflow_step.id")), + Column("dataset_id", Integer, ForeignKey("history_dataset_association.id"), index=True), + Column("workflow_output_id", Integer, ForeignKey("workflow_output.id")), + ) + + workflow_invocation_output_dataset_collection_association_table = Table( + "workflow_invocation_output_dataset_collection_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True), + Column("workflow_step_id", Integer, ForeignKey("workflow_step.id")), + Column("dataset_collection_id", Integer, ForeignKey("history_dataset_collection_association.id"), index=True), + Column("workflow_output_id", Integer, ForeignKey("workflow_output.id")), + ) + + workflow_invocation_step_output_dataset_association_table = Table( + "workflow_invocation_step_output_dataset_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True), + Column("dataset_id", Integer, ForeignKey("history_dataset_association.id"), index=True), + Column("output_name", String(255), nullable=True), + ) + + workflow_invocation_step_output_dataset_collection_association_table = Table( + "workflow_invocation_step_output_dataset_collection_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True), + Column("workflow_step_id", Integer, ForeignKey("workflow_step.id")), + Column("dataset_collection_id", Integer, ForeignKey("history_dataset_collection_association.id"), index=True), + Column("output_name", String(255), nullable=True), + ) + + workflow_invocation_step_table = Table( + "workflow_invocation_step", metadata, + Column("id", Integer, primary_key=True), + Column("create_time", DateTime, default=now), + Column("update_time", DateTime, default=now, onupdate=now), + Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True, nullable=False), + Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True, nullable=False), + Column("action", JSONType, nullable=True), + Column("state", TrimmedString(64), default="new"), + ) + + workflow_invocation_step_job_association_table = Table( + "workflow_invocation_step_job_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True, nullable=False), + Column("order_index", Integer, nullable=True), + Column("job_id", Integer, ForeignKey("job.id"), index=True, nullable=False), + ) + + tables = OrderedDict() + tables["workflow_invocation_step"] = workflow_invocation_step_table + tables["workflow_invocation_output_dataset_association"] = workflow_invocation_output_dataset_association_table + tables["workflow_invocation_output_dataset_collection_association"] = workflow_invocation_output_dataset_collection_association_table + tables["workflow_invocation_step_output_dataset_association"] = workflow_invocation_step_output_dataset_association_table + tables["workflow_invocation_step_output_dataset_collection_association"] = workflow_invocation_step_output_dataset_collection_association_table + tables["workflow_invocation_step_job_association"] = workflow_invocation_step_job_association_table + + return tables + + +def upgrade(migrate_engine): + metadata.bind = migrate_engine + print(__doc__) + + LegacyWorkflowInvocationStep_table = Table("workflow_invocation_step", metadata, autoload=True) + ExistingWorkflowInvocation_table = Table("workflow_invocation", metadata, autoload=True) + + cons = ForeignKeyConstraint([LegacyWorkflowInvocationStep_table.c.workflow_invocation_id], [ExistingWorkflowInvocation_table.c.id]) + cons.drop() + + for index in LegacyWorkflowInvocationStep_table.indexes: + index.drop() + + LegacyWorkflowInvocationStep_table.rename("workflow_invocation_step_premigrate135") + # Try to deregister that workflow_invocation_step to work around some caching problems + # it seems. + LegacyWorkflowInvocationStep_table.deregister() + metadata._remove_table("workflow_invocation_step", metadata.schema) + + metadata.reflect() + tables = get_new_tables() + for table in tables.values(): + __create(table) + + def nextval(table, col='id'): + if migrate_engine.name in ['postgres', 'postgresql']: + return "nextval('%s_%s_seq')" % (table, col) + elif migrate_engine.name in ['mysql', 'sqlite']: + return "null" + else: + raise Exception("Unhandled database type") + + # Skips action - since I can't aggregate (sql needs a RANDOM(col)) that and it is only used by optional + # beta extensions. + cmd = \ + "INSERT INTO workflow_invocation_step " + \ + "(id, create_time, update_time, workflow_invocation_id, workflow_step_id, action, state)" + \ + "SELECT " + \ + nextval('workflow_invocation_step') + " AS id, " \ + "MIN(create_time) AS create_time, " + \ + "MAX(update_time) AS update_time, " + \ + "workflow_invocation_step_premigrate135.workflow_invocation_id AS workflow_invocation_id, " +\ + "workflow_invocation_step_premigrate135.workflow_step_id AS workflow_step_id, " + \ + "NULL AS action, " + \ + "'scheduled' AS state " + \ + "FROM workflow_invocation_step_premigrate135 " + \ + "WHERE workflow_invocation_step_premigrate135.workflow_step_id IS NOT NULL AND workflow_invocation_step_premigrate135.workflow_invocation_id IS NOT NULL " +\ + "GROUP BY workflow_invocation_step_premigrate135.workflow_invocation_id, workflow_invocation_step_premigrate135.workflow_step_id " + \ + "" + migrate_engine.execute(cmd) + + cmd = \ + "INSERT INTO workflow_invocation_step_job_association " + \ + "(id, workflow_invocation_step_id, order_index, job_id) " + \ + "SELECT " + \ + nextval('workflow_invocation_step_job_association') + " AS id, " \ + "workflow_invocation_step.id AS workflow_invocation_step_id, " + \ + "NULL AS order_index, " + \ + "job_id AS job_id " + \ + "FROM workflow_invocation_step_premigrate135 " + \ + "LEFT JOIN workflow_invocation_step on (" + \ + " workflow_invocation_step.workflow_invocation_id = workflow_invocation_step_premigrate135.workflow_invocation_id " + \ + " AND workflow_invocation_step.workflow_step_id = workflow_invocation_step_premigrate135.workflow_step_id" + \ + ") " + \ + "WHERE job_id is not NULL " + migrate_engine.execute(cmd) + + +def downgrade(migrate_engine): + metadata.bind = migrate_engine + metadata.reflect() + + tables = get_new_tables() + for table in tables.values(): + __drop(table) + + # Drop new workflow invocation step and job association table and restore legacy data. + LegacyWorkflowInvocationStep_table = Table("workflow_invocation_step_premigrate135", metadata, autoload=True) + LegacyWorkflowInvocationStep_table.rename("workflow_invocation_step") + + +def __create(table): + try: + table.create() + except Exception: + log.exception("Creating %s table failed.", table.name) + + +def __drop(table): + try: + table.drop() + except Exception: + log.exception("Dropping %s table failed.", table.name) diff --git a/lib/galaxy/tools/execute.py b/lib/galaxy/tools/execute.py index c24a57af1c0..e6d37538110 100644 --- a/lib/galaxy/tools/execute.py +++ b/lib/galaxy/tools/execute.py @@ -18,7 +18,13 @@ log = logging.getLogger(__name__) EXECUTION_SUCCESS_MESSAGE = "Tool [%s] created job [%s] %s" -def execute(trans, tool, param_combinations, history, rerun_remap_job_id=None, collection_info=None, workflow_invocation_uuid=None): +class PartialJobExecution(Exception): + + def __init__(self, jobs): + self.jobs = jobs + + +def execute(trans, tool, param_combinations, history, rerun_remap_job_id=None, collection_info=None, workflow_invocation_uuid=None, invocation_step=None, max_num_jobs=None): """ Execute a tool and return object containing summary (output data, number of failures, etc...). @@ -28,6 +34,8 @@ def execute(trans, tool, param_combinations, history, rerun_remap_job_id=None, c app = trans.app execution_cache = ToolExecutionCache(trans) + new_jobs = [] + def execute_single_job(params): job_timer = ExecutionTimer() if workflow_invocation_uuid: @@ -41,6 +49,7 @@ def execute(trans, tool, param_combinations, history, rerun_remap_job_id=None, c message = EXECUTION_SUCCESS_MESSAGE % (tool.id, job.id, job_timer) log.debug(message) execution_tracker.record_success(job, result) + new_jobs.append(job) else: execution_tracker.record_error(result) @@ -59,11 +68,27 @@ def execute(trans, tool, param_combinations, history, rerun_remap_job_id=None, c history ) + if invocation_step: + execution_tracker.recover_successful_jobs(invocation_step) + + previously_executed_jobs_count = len(execution_tracker.successful_jobs) job_count = len(execution_tracker.param_combinations) - if job_count < burst_at or burst_threads < 2: - for params in execution_tracker.param_combinations: - execute_single_job(params) + + jobs_executed = 0 + has_remaining_jobs = False + + if (job_count < burst_at or burst_threads < 2): + for index, params in enumerate(execution_tracker.param_combinations): + if index < previously_executed_jobs_count: + continue + elif max_num_jobs and jobs_executed >= max_num_jobs: + has_remaining_jobs = True + break + else: + execute_single_job(params) + jobs_executed += 1 else: + # TODO: re-record success... q = Queue() def worker(): @@ -77,11 +102,21 @@ def execute(trans, tool, param_combinations, history, rerun_remap_job_id=None, c t.daemon = True t.start() - for params in execution_tracker.param_combinations: - q.put(params) + for index, params in enumerate(execution_tracker.param_combinations): + if index < previously_executed_jobs_count: + continue + elif max_num_jobs and jobs_executed >= max_num_jobs: + has_remaining_jobs = True + break + else: + q.put(params) + jobs_executed += 1 q.join() + if has_remaining_jobs: + raise PartialJobExecution(new_jobs) + log.debug("Executed %d job(s) for tool %s request: %s" % (job_count, tool.id, all_jobs_timer)) if collection_info: history = history or tool.get_default_history_by_trans(trans) @@ -110,6 +145,17 @@ class ToolExecutionTracker(object): self.outputs_by_output_name = collections.defaultdict(list) self.implicit_collections = {} + def recover_successful_jobs(self, invocation_step): + # TODO: Optimize away the need to do this - we should just be dealing with IDs + # and such and we shouldn't fetch them until the very end when we need them to create + # collections. + for job_assoc in invocation_step.jobs: + job = job_assoc.job + for job_output in job.output_datasets: + self.outputs_by_output_name[job_output.name].append(job_output.dataset) + for job_output in job.output_dataset_collections: + self.outputs_by_output_name[job_output.name].append(job_output.dataset_collection) + def record_success(self, job, outputs): self.successful_jobs.append(job) self.output_datasets.extend(outputs) diff --git a/lib/galaxy/workflow/modules.py b/lib/galaxy/workflow/modules.py index d5fcbf641cf..7abc55636a5 100644 --- a/lib/galaxy/workflow/modules.py +++ b/lib/galaxy/workflow/modules.py @@ -21,7 +21,7 @@ from galaxy.tools import ( DefaultToolState, ToolInputsNotReadyException ) -from galaxy.tools.execute import execute +from galaxy.tools.execute import execute, PartialJobExecution from galaxy.tools.parameters import ( check_param, params_to_incoming, @@ -214,10 +214,14 @@ class WorkflowModule(object): state.decode(runtime_state, Bunch(inputs=self.get_runtime_inputs()), self.trans.app) return state - def execute(self, trans, progress, invocation, step): - """ Execute the given workflow step in the given workflow invocation. + def execute(self, trans, progress, invocation_step): + """ Execute the given workflow invocation step. + Use the supplied workflow progress object to track outputs, find - inputs, etc... + inputs, etc.... + + Return jobs created and a boolean indicating if there are additional + jobs to create on subsequent workflow scheduling tests. """ raise TypeError("Abstract method") @@ -230,11 +234,19 @@ class WorkflowModule(object): """ raise exceptions.RequestParameterInvalidException("Attempting to perform invocation step action on module that does not support actions.") - def recover_mapping(self, step, step_invocations, progress): + def recover_mapping(self, invocation_step, progress): """ Re-populate progress object with information about connections - from previously executed steps recorded via step_invocations. + from previously executed steps recorded via invocation_steps. """ - raise TypeError("Abstract method") + outputs = {} + + for output_dataset_assoc in invocation_step.output_datasets: + outputs[output_dataset_assoc.output_name] = output_dataset_assoc.dataset + + for output_dataset_collection_assoc in invocation_step.output_dataset_collections: + outputs[output_dataset_collection_assoc.output_name] = output_dataset_collection_assoc.dataset_collection + + progress.set_step_outputs(invocation_step, outputs, already_persisted=True) class SubWorkflowModule(WorkflowModule): @@ -320,11 +332,12 @@ class SubWorkflowModule(WorkflowModule): def get_content_id(self): return self.trans.security.encode_id(self.subworkflow.id) - def execute(self, trans, progress, invocation, step): + def execute(self, trans, progress, invocation_step): """ Execute the given workflow step in the given workflow invocation. Use the supplied workflow progress object to track outputs, find inputs, etc... """ + step = invocation_step.workflow_step subworkflow_invoker = progress.subworkflow_invoker(trans, step) subworkflow_invoker.invoke() subworkflow = subworkflow_invoker.workflow @@ -334,7 +347,7 @@ class SubWorkflowModule(WorkflowModule): workflow_output_label = workflow_output.label or "%s:%s" % (step.order_index, workflow_output.output_name) replacement = subworkflow_progress.get_replacement_workflow_output(workflow_output) outputs[workflow_output_label] = replacement - progress.set_step_outputs(step, outputs) + progress.set_step_outputs(invocation_step, outputs) return None def get_runtime_state(self): @@ -353,8 +366,10 @@ class InputModule(WorkflowModule): def get_data_inputs(self): return [] - def execute(self, trans, progress, invocation, step): - job, step_outputs = None, dict(output=step.state.inputs['input']) + def execute(self, trans, progress, invocation_step): + invocation = invocation_step.workflow_invocation + step = invocation_step.workflow_step + step_outputs = dict(output=step.state.inputs['input']) # Web controller may set copy_inputs_to_history, API controller always sets # inputs. @@ -378,11 +393,10 @@ class InputModule(WorkflowModule): content = next(iter(step_outputs.values())) if content: invocation.add_input(content, step.id) - progress.set_outputs_for_input(step, step_outputs) - return job + progress.set_outputs_for_input(invocation_step, step_outputs) - def recover_mapping(self, step, step_invocations, progress): - progress.set_outputs_for_input(step) + def recover_mapping(self, invocation_step, progress): + progress.set_outputs_for_input(invocation_step) class InputDataModule(InputModule): @@ -489,10 +503,10 @@ class InputParameterModule(WorkflowModule): def get_data_inputs(self): return [] - def execute(self, trans, progress, invocation, step): - job, step_outputs = None, dict(output=step.state.inputs['input']) - progress.set_outputs_for_input(step, step_outputs) - return job + def execute(self, trans, progress, invocation_step): + step = invocation_step.workflow_step + step_outputs = dict(output=step.state.inputs['input']) + progress.set_outputs_for_input(invocation_step, step_outputs) class PauseModule(WorkflowModule): @@ -520,18 +534,18 @@ class PauseModule(WorkflowModule): state.inputs = dict() return state - def execute(self, trans, progress, invocation, step): + def execute(self, trans, progress, invocation_step): + step = invocation_step.workflow_step progress.mark_step_outputs_delayed(step, why="executing pause step") - return None - def recover_mapping(self, step, step_invocations, progress): - if step_invocations: - step_invocation = step_invocations[0] - action = step_invocation.action + def recover_mapping(self, invocation_step, progress): + if invocation_step: + step = invocation_step.workflow_step + action = invocation_step.action if action: connection = step.input_connections_by_name["input"][0] replacement = progress.replacement_for_connection(connection) - progress.set_step_outputs(step, {'output': replacement}) + progress.set_step_outputs(invocation_step, {'output': replacement}) return elif action is False: raise CancelWorkflowEvaluation() @@ -790,7 +804,9 @@ class ToolModule(WorkflowModule): else: raise ToolMissingException("Tool %s missing. Cannot recover runtime state." % self.tool_id) - def execute(self, trans, progress, invocation, step): + def execute(self, trans, progress, invocation_step): + invocation = invocation_step.workflow_invocation + step = invocation_step.workflow_step tool = trans.app.toolbox.get_tool(step.tool_id, tool_version=step.tool_version) tool_state = step.state # Not strictly needed - but keep Tool state clean by stripping runtime @@ -855,6 +871,9 @@ class ToolModule(WorkflowModule): param_combinations.append(execution_state.inputs) + # Will be set if only a subset of required jobs have been scheduled and the + # workflow should be delayed. + partial_jobs = None try: execution_tracker = execute( trans=self.trans, @@ -862,50 +881,49 @@ class ToolModule(WorkflowModule): param_combinations=param_combinations, history=invocation.history, collection_info=collection_info, - workflow_invocation_uuid=invocation.uuid.hex + workflow_invocation_uuid=invocation.uuid.hex, + invocation_step=invocation_step, ) + except PartialJobExecution as p: + partial_jobs = p.jobs except ToolInputsNotReadyException: delayed_why = "tool [%s] inputs are not ready, this special tool requires inputs to be ready" % tool.id raise DelayedWorkflowEvaluation(why=delayed_why) - if collection_info: - step_outputs = dict(execution_tracker.implicit_collections) + if partial_jobs is None: + if collection_info: + step_outputs = dict(execution_tracker.implicit_collections) + else: + step_outputs = dict(execution_tracker.output_datasets) + step_outputs.update(execution_tracker.output_collections) + progress.set_step_outputs(invocation_step, step_outputs) + jobs = execution_tracker.successful_jobs else: - step_outputs = dict(execution_tracker.output_datasets) - step_outputs.update(execution_tracker.output_collections) - progress.set_step_outputs(step, step_outputs) - jobs = execution_tracker.successful_jobs + jobs = partial_jobs + for job in jobs: self._handle_post_job_actions(step, job, invocation.replacement_dict) + if execution_tracker.execution_errors: failed_count = len(execution_tracker.execution_errors) success_count = len(execution_tracker.successful_jobs) all_count = failed_count + success_count message = "Failed to create %d out of %s job(s) for workflow step." % (failed_count, all_count) raise Exception(message) - return jobs - def recover_mapping(self, step, step_invocations, progress): - # Grab a job representing this invocation - for normal workflows - # there will be just one job but if this step was mapped over there - # may be many. - job_0 = step_invocations[0].job + complete = partial_jobs is None + return jobs, complete + def recover_mapping(self, invocation_step, progress): outputs = {} - for job_output in job_0.output_datasets: - replacement_name = job_output.name - replacement_value = job_output.dataset - # If was a mapping step, grab the output mapped collection for - # replacement instead. - if replacement_value.hidden_beneath_collection_instance: - replacement_value = replacement_value.hidden_beneath_collection_instance - outputs[replacement_name] = replacement_value - for job_output_collection in job_0.output_dataset_collection_instances: - replacement_name = job_output_collection.name - replacement_value = job_output_collection.dataset_collection_instance - outputs[replacement_name] = replacement_value - progress.set_step_outputs(step, outputs) + for output_dataset_assoc in invocation_step.output_datasets: + outputs[output_dataset_assoc.output_name] = output_dataset_assoc.dataset + + for output_dataset_collection_assoc in invocation_step.output_dataset_collections: + outputs[output_dataset_collection_assoc.output_name] = output_dataset_collection_assoc.dataset_collection + + progress.set_step_outputs(invocation_step, outputs) def _find_collections_to_match(self, tool, progress, step): collections_to_match = matching.CollectionsToMatch() diff --git a/lib/galaxy/workflow/run.py b/lib/galaxy/workflow/run.py index 0f6fd041d04..75222d5a46d 100644 --- a/lib/galaxy/workflow/run.py +++ b/lib/galaxy/workflow/run.py @@ -162,24 +162,46 @@ class WorkflowInvoker(object): remaining_steps = self.progress.remaining_steps() delayed_steps = False - for step in remaining_steps: + for (step, workflow_invocation_step) in remaining_steps: step_delayed = False step_timer = ExecutionTimer() jobs = None try: self.__check_implicitly_dependent_steps(step) - # TODO: step may fail to invoke, do something about that. - jobs = self._invoke_step(step) - for job in (util.listify(jobs) or [None]): - # Record invocation + if not workflow_invocation_step: workflow_invocation_step = model.WorkflowInvocationStep() workflow_invocation_step.workflow_invocation = workflow_invocation workflow_invocation_step.workflow_step = step + workflow_invocation_step.state = 'new' + previously_executed_jobs_count = 0 + + workflow_invocation.steps.append(workflow_invocation_step) + else: + previously_executed_jobs_count = len(workflow_invocation_step.jobs) + + jobs_or_none = self._invoke_step(workflow_invocation_step) + if jobs_or_none: + jobs, complete = jobs_or_none + else: + jobs, complete = [], True + + for job in (util.listify(jobs) or []): + job_assoc = model.WorkflowInvocationStepJobAssociation() + job_assoc.index = previously_executed_jobs_count + job_assoc.workflow_invocation_step = workflow_invocation_step # Job may not be generated in this thread if bursting is enabled # https://github.com/galaxyproject/galaxy/issues/2259 - if job: - workflow_invocation_step.job_id = job.id + job_assoc.job_id = job.id + + previously_executed_jobs_count += 1 + + if not complete: + step_delayed = delayed_steps = True + workflow_invocation_step.state = 'ready' + self.progress.mark_step_outputs_delayed(step, why="Not all jobs scheduled for state.") + else: + workflow_invocation_step.state = 'scheduled' except modules.DelayedWorkflowEvaluation as de: step_delayed = delayed_steps = True self.progress.mark_step_outputs_delayed(step, why=de.why) @@ -218,15 +240,19 @@ class WorkflowInvoker(object): self.__check_implicitly_dependent_step(output_id) def __check_implicitly_dependent_step(self, output_id): - step_invocations = self.workflow_invocation.step_invocations_for_step_id(output_id) + step_invocation = self.workflow_invocation.step_invocation_for_step_id(output_id) # No steps created yet - have to delay evaluation. - if not step_invocations: + if not step_invocation: delayed_why = "depends on step [%s] but that step has not been invoked yet" % output_id raise modules.DelayedWorkflowEvaluation(why=delayed_why) - for step_invocation in step_invocations: - job = step_invocation.job + if step_invocation.state != 'scheduled': + delayed_why = "depends on step [%s] job has not finished scheduling yet" % output_id + raise modules.DelayedWorkflowEvaluation(delayed_why) + + for job_assoc in step_invocation.jobs: + job = job_assoc.job if job: # At least one job in incomplete. if not job.finished: @@ -241,9 +267,9 @@ class WorkflowInvoker(object): # pause steps. pass - def _invoke_step(self, step): - jobs = step.module.execute(self.trans, self.progress, self.workflow_invocation, step) - return jobs + def _invoke_step(self, invocation_step): + jobs_or_none = invocation_step.workflow_step.module.execute(self.trans, self.progress, invocation_step) + return jobs_or_none STEP_OUTPUT_DELAYED = object() @@ -256,11 +282,19 @@ class WorkflowProgress(object): self.module_injector = module_injector self.workflow_invocation = workflow_invocation self.inputs_by_step_id = inputs_by_step_id + self.jobs_per_scheduling_iteration = 1 + + @property + def maximum_jobs_to_schedule(self): + return 1 def remaining_steps(self): # Previously computed and persisted step states. step_states = self.workflow_invocation.step_states_by_step_id() steps = self.workflow_invocation.workflow.steps + + # TODO: Wouldn't a generator be much better here so we don't have to reason about + # steps we are no where near ready to schedule? remaining_steps = [] step_invocations_by_id = self.workflow_invocation.step_invocations_by_step_id() for step in steps: @@ -274,11 +308,11 @@ class WorkflowProgress(object): runtime_state = step_states[step_id].value step.state = step.module.decode_runtime_state(runtime_state) - invocation_steps = step_invocations_by_id.get(step_id, None) - if invocation_steps: - self._recover_mapping(step, invocation_steps) + invocation_step = step_invocations_by_id.get(step_id, None) + if invocation_step and invocation_step.state == 'scheduled': + self._recover_mapping(invocation_step) else: - remaining_steps.append(step) + remaining_steps.append((step, invocation_step)) return remaining_steps def replacement_for_tool_input(self, step, input, prefixed_name): @@ -340,7 +374,9 @@ class WorkflowProgress(object): output_name = workflow_output.output_name return self.outputs[step.id][output_name] - def set_outputs_for_input(self, step, outputs=None): + def set_outputs_for_input(self, invocation_step, outputs=None): + step = invocation_step.workflow_step + if outputs is None: outputs = {} @@ -352,10 +388,34 @@ class WorkflowProgress(object): raise ValueError(message) outputs['output'] = self.inputs_by_step_id[step_id] - self.set_step_outputs(step, outputs) + self.set_step_outputs(invocation_step, outputs) - def set_step_outputs(self, step, outputs): + def set_step_outputs(self, invocation_step, outputs, already_persisted=False): + step = invocation_step.workflow_step self.outputs[step.id] = outputs + if not already_persisted: + for output_name, output_object in outputs.items(): + if hasattr(output_object, "history_content_type"): + invocation_step.add_output(output_name, output_object) + else: + # This is a problem, this non-data, non-collection output + # won't be recovered on a subsequent workflow scheduling + # iteration. This seems to have been a pre-existing problem + # prior to #4584 though. + pass + for workflow_output in step.workflow_outputs: + output_name = workflow_output.output_name + if output_name not in outputs: + raise KeyError("Failed to find [%s] in step outputs [%s]" % (output_name, outputs)) + output = outputs[output_name] + self._record_workflow_output( + step, + workflow_output, + output=output, + ) + + def _record_workflow_output(self, step, workflow_output, output): + self.workflow_invocation.add_output(workflow_output, step, output) def mark_step_outputs_delayed(self, step, why=None): if why: @@ -414,11 +474,11 @@ class WorkflowProgress(object): self.module_injector, ) - def _recover_mapping(self, step, step_invocations): + def _recover_mapping(self, step_invocation): try: - step.module.recover_mapping(step, step_invocations, self) + step_invocation.workflow_step.module.recover_mapping(step_invocation, self) except modules.DelayedWorkflowEvaluation as de: - self.mark_step_outputs_delayed(step, de.why) + self.mark_step_outputs_delayed(step_invocation.workflow_step, de.why) __all__ = ('invoke', 'WorkflowRunConfig') diff --git a/test/api/test_workflows.py b/test/api/test_workflows.py index f875c2c9fa9..214692f1021 100644 --- a/test/api/test_workflows.py +++ b/test/api/test_workflows.py @@ -963,6 +963,77 @@ test_data: time.sleep(5) self.dataset_populator.wait_for_history(history_id, assert_ok=True) + def test_workflow_output_dataset(self): + history_id = self.dataset_populator.new_history() + summary = self._run_jobs(""" +class: GalaxyWorkflow +inputs: + - id: input1 +outputs: + - id: wf_output_1 + source: first_cat#out_file1 +steps: + - tool_id: cat1 + label: first_cat + state: + input1: + $link: input1 + +test_data: + input1: "hello world" +""", history_id=history_id) + workflow_id = summary.workflow_id + invocation_id = summary.invocation_id + invocation_response = self._get("workflows/%s/invocations/%s" % (workflow_id, invocation_id)) + self._assert_status_code_is(invocation_response, 200) + invocation = invocation_response.json() + self._assert_has_keys(invocation , "id", "outputs", "output_collections") + assert len(invocation["output_collections"]) == 0 + assert len(invocation["outputs"]) == 1 + output_content = self.dataset_populator.get_history_dataset_content(history_id, dataset_id=invocation["outputs"]["wf_output_1"]["id"]) + assert "hello world" == output_content.strip() + + def test_workflow_output_dataset_collection(self): + history_id = self.dataset_populator.new_history() + summary = self._run_jobs(""" +class: GalaxyWorkflow +inputs: + - id: input1 + type: data_collection_input + collection_type: list +outputs: + - id: wf_output_1 + source: first_cat#out_file1 +steps: + - tool_id: cat + label: first_cat + state: + input1: + $link: input1 +test_data: + input1: + type: list + name: the_dataset_list + elements: + - identifier: el1 + value: 1.fastq + type: File +""", history_id=history_id) + workflow_id = summary.workflow_id + invocation_id = summary.invocation_id + invocation_response = self._get("workflows/%s/invocations/%s" % (workflow_id, invocation_id)) + self._assert_status_code_is(invocation_response, 200) + invocation = invocation_response.json() + self._assert_has_keys(invocation , "id", "outputs", "output_collections") + assert len(invocation["output_collections"]) == 1 + assert len(invocation["outputs"]) == 0 + output_content = self.dataset_populator.get_history_collection_details(history_id, content_id=invocation["output_collections"]["wf_output_1"]["id"]) + self._assert_has_keys(output_content , "id", "elements") + elements = output_content["elements"] + assert len(elements) == 1 + elements0 = elements[0] + assert elements0["element_identifier"] == "el1" + @skip_without_tool("cat") def test_cancel_new_workflow_when_history_deleted(self): with self.dataset_populator.test_history() as history_id: diff --git a/test/base/populators.py b/test/base/populators.py index e382ace1ead..b64b57ff526 100644 --- a/test/base/populators.py +++ b/test/base/populators.py @@ -282,6 +282,8 @@ class BaseDatasetPopulator(object): # the last dataset in the history will be fetched. if "dataset_id" in kwds: history_content_id = kwds["dataset_id"] + elif "content_id" in kwds: + history_content_id = kwds["content_id"] elif "dataset" in kwds: history_content_id = kwds["dataset"]["id"] else: diff --git a/test/unit/workflows/test_workflow_progress.py b/test/unit/workflows/test_workflow_progress.py index 21cb2341414..3991cb393a4 100644 --- a/test/unit/workflows/test_workflow_progress.py +++ b/test/unit/workflows/test_workflow_progress.py @@ -72,13 +72,17 @@ class WorkflowProgressTestCase(unittest.TestCase): self.invocation, self.inputs_by_step_id, MockModuleInjector(self.progress) ) - def _set_previous_progress(self, outputs_dict): - for step_id, step_value in outputs_dict.items(): + def _set_previous_progress(self, outputs): + for i, (step_id, step_value) in enumerate(outputs): if step_value is not UNSCHEDULED_STEP: self.progress[step_id] = step_value workflow_invocation_step = model.WorkflowInvocationStep() workflow_invocation_step.workflow_step_id = step_id + workflow_invocation_step.state = 'scheduled' + workflow_invocation_step.workflow_step = self._step(i) + self.assertEqual(step_id, self._step(i).id) + # workflow_invocation_step.workflow_invocation = self.invocation self.invocation.steps.append(workflow_invocation_step) workflow_invocation_step_state = model.WorkflowRequestStepState() @@ -89,13 +93,21 @@ class WorkflowProgressTestCase(unittest.TestCase): def _step(self, index): return self.invocation.workflow.steps[index] + def _invocation_step(self, index): + if index < len(self.invocation.steps): + return self.invocation.steps[index] + else: + workflow_invocation_step = model.WorkflowInvocationStep() + workflow_invocation_step.workflow_step = self._step(index) + return workflow_invocation_step + def test_connect_data_input(self): self._setup_workflow(TEST_WORKFLOW_YAML) hda = model.HistoryDatasetAssociation() self.inputs_by_step_id = {100: hda} progress = self._new_workflow_progress() - progress.set_outputs_for_input(self._step(0)) + progress.set_outputs_for_input(self._invocation_step(0)) conn = model.WorkflowStepConnection() conn.output_name = "output" @@ -108,7 +120,7 @@ class WorkflowProgressTestCase(unittest.TestCase): self.inputs_by_step_id = {100: hda} progress = self._new_workflow_progress() - progress.set_outputs_for_input(self._step(0)) + progress.set_outputs_for_input(self._invocation_step(0)) replacement = progress.replacement_for_tool_input(self._step(2), MockInput(), "input1") assert replacement is hda @@ -118,7 +130,7 @@ class WorkflowProgressTestCase(unittest.TestCase): hda = model.HistoryDatasetAssociation() progress = self._new_workflow_progress() - progress.set_step_outputs(self._step(2), {"out1": hda}) + progress.set_step_outputs(self._invocation_step(2), {"out1": hda}) conn = model.WorkflowStepConnection() conn.output_name = "out1" @@ -128,17 +140,18 @@ class WorkflowProgressTestCase(unittest.TestCase): def test_remaining_steps_with_progress(self): self._setup_workflow(TEST_WORKFLOW_YAML) hda3 = model.HistoryDatasetAssociation() - self._set_previous_progress({ - 100: {"output": model.HistoryDatasetAssociation()}, - 101: {"output": model.HistoryDatasetAssociation()}, - 102: {"out_file1": hda3}, - 103: {"out_file1": model.HistoryDatasetAssociation()}, - 104: UNSCHEDULED_STEP, - }) + self._set_previous_progress([ + (100, {"output": model.HistoryDatasetAssociation()}), + (101, {"output": model.HistoryDatasetAssociation()}), + (102, {"out_file1": hda3}), + (103, {"out_file1": model.HistoryDatasetAssociation()}), + (104, UNSCHEDULED_STEP), + ]) progress = self._new_workflow_progress() steps = progress.remaining_steps() - assert len(steps) == 1 - assert steps[0] is self.invocation.workflow.steps[4] + assert len(steps) == 1, steps + step, invocation_step = steps[0] + assert step is self.invocation.workflow.steps[4] replacement = progress.replacement_for_tool_input(self._step(4), MockInput(), "input1") assert replacement is hda3 @@ -151,21 +164,27 @@ class WorkflowProgressTestCase(unittest.TestCase): def test_subworkflow_progress(self): self._setup_workflow(TEST_SUBWORKFLOW_YAML) hda = model.HistoryDatasetAssociation() - self._set_previous_progress({ - 100: {"output": hda}, - 101: UNSCHEDULED_STEP, - }) + self._set_previous_progress([ + (100, {"output": hda}), + (101, UNSCHEDULED_STEP), + ]) self.invocation.create_subworkflow_invocation_for_step( self.invocation.workflow.step_by_index(1) ) progress = self._new_workflow_progress() remaining_steps = progress.remaining_steps() - subworkflow_step = remaining_steps[0] + (subworkflow_step, subworkflow_invocation_step) = remaining_steps[0] subworkflow_progress = progress.subworkflow_progress(subworkflow_step) subworkflow = subworkflow_step.subworkflow assert subworkflow_progress.workflow_invocation.workflow == subworkflow + subworkflow_input_step = subworkflow.step_by_index(0) - subworkflow_progress.set_outputs_for_input(subworkflow_input_step) + subworkflow_invocation_step = model.WorkflowInvocationStep() + subworkflow_invocation_step.workflow_step_id = subworkflow_input_step.id + subworkflow_invocation_step.state = 'new' + subworkflow_invocation_step.workflow_step = subworkflow_input_step + + subworkflow_progress.set_outputs_for_input(subworkflow_invocation_step) subworkflow_cat_step = subworkflow.step_by_index(1) @@ -200,7 +219,7 @@ class MockModule(object): def decode_runtime_state(self, runtime_state): return True - def recover_mapping(self, step, step_invocations, progress): - step_id = step.id + def recover_mapping(self, invocation_step, progress): + step_id = invocation_step.workflow_step.id if step_id in self.progress: - progress.set_step_outputs(step, self.progress[step_id]) + progress.set_step_outputs(invocation_step, self.progress[step_id]) From 64835c1e72eea545269260256c7a238a4e5ea7fe Mon Sep 17 00:00:00 2001 From: John Chilton Date: Tue, 8 Aug 2017 13:47:28 -0400 Subject: [PATCH 08/22] Three new worklfow test cases to verify mapping over workflows is possible. - Simple mapping over an input dataset. - Mapping that produces nested collections. - Mapping over subworkflows. --- test/api/test_workflows.py | 138 +++++++++++++++++++++++++++++++++++-- 1 file changed, 132 insertions(+), 6 deletions(-) diff --git a/test/api/test_workflows.py b/test/api/test_workflows.py index 214692f1021..b3501adc9a2 100644 --- a/test/api/test_workflows.py +++ b/test/api/test_workflows.py @@ -24,6 +24,9 @@ SIMPLE_NESTED_WORKFLOW_YAML = """ class: GalaxyWorkflow inputs: - id: outer_input +outputs: + - id: outer_output + source: second_cat#out_file1 steps: - tool_id: cat1 label: first_cat @@ -58,11 +61,6 @@ steps: queries: - input2: $link: nested_workflow#workflow_output - -test_data: - outer_input: - value: 1.bed - type: File """ @@ -861,7 +859,14 @@ test_data: def test_run_subworkflow_simple(self): history_id = self.dataset_populator.new_history() - self._run_jobs(SIMPLE_NESTED_WORKFLOW_YAML, history_id=history_id) + workflow_run_description = """%s + +test_data: + outer_input: + value: 1.bed + type: File +""" % SIMPLE_NESTED_WORKFLOW_YAML + self._run_jobs(workflow_run_description, history_id=history_id) content = self.dataset_populator.get_history_dataset_content(history_id) self.assertEqual("chr5\t131424298\t131424460\tCCDS4149.1_cds_0_0_chr5_131424299_f\t0\t+\nchr5\t131424298\t131424460\tCCDS4149.1_cds_0_0_chr5_131424299_f\t0\t+\n", content) @@ -1029,11 +1034,132 @@ test_data: assert len(invocation["outputs"]) == 0 output_content = self.dataset_populator.get_history_collection_details(history_id, content_id=invocation["output_collections"]["wf_output_1"]["id"]) self._assert_has_keys(output_content , "id", "elements") + assert output_content["collection_type"] == "list" elements = output_content["elements"] assert len(elements) == 1 elements0 = elements[0] assert elements0["element_identifier"] == "el1" + def test_worklfow_input_mapping(self): + history_id = self.dataset_populator.new_history() + summary = self._run_jobs(""" +class: GalaxyWorkflow +inputs: + - id: input1 +outputs: + - id: wf_output_1 + source: first_cat#out_file1 +steps: + - tool_id: cat + label: first_cat + state: + input1: + $link: input1 +test_data: + input1: + type: list + name: the_dataset_list + elements: + - identifier: el1 + value: 1.fastq + type: File + - identifier: el2 + value: 1.fastq + type: File +""", history_id=history_id) + workflow_id = summary.workflow_id + invocation_id = summary.invocation_id + invocation_response = self._get("workflows/%s/invocations/%s" % (workflow_id, invocation_id)) + self._assert_status_code_is(invocation_response, 200) + invocation = invocation_response.json() + self._assert_has_keys(invocation , "id", "outputs", "output_collections") + assert len(invocation["output_collections"]) == 1 + assert len(invocation["outputs"]) == 0 + output_content = self.dataset_populator.get_history_collection_details(history_id, content_id=invocation["output_collections"]["wf_output_1"]["id"]) + self._assert_has_keys(output_content , "id", "elements") + elements = output_content["elements"] + assert len(elements) == 2 + elements0 = elements[0] + assert elements0["element_identifier"] == "el1" + + @skip_without_tool("collection_creates_pair") + def test_workflow_run_input_mapping_with_output_collections(self): + history_id = self.dataset_populator.new_history() + summary = self._run_jobs(""" +class: GalaxyWorkflow +outputs: + - id: wf_output_1 + source: split_up#paired_output +steps: + - label: text_input + type: input + - label: split_up + tool_id: collection_creates_pair + state: + input1: + $link: text_input +test_data: + text_input: + type: list + name: the_dataset_list + elements: + - identifier: el1 + value: 1.fastq + type: File + - identifier: el2 + value: 1.fastq + type: File +""", history_id=history_id) + workflow_id = summary.workflow_id + invocation_id = summary.invocation_id + invocation_response = self._get("workflows/%s/invocations/%s" % (workflow_id, invocation_id)) + self._assert_status_code_is(invocation_response, 200) + invocation = invocation_response.json() + self._assert_has_keys(invocation , "id", "outputs", "output_collections") + assert len(invocation["output_collections"]) == 1 + assert len(invocation["outputs"]) == 0 + output_content = self.dataset_populator.get_history_collection_details(history_id, content_id=invocation["output_collections"]["wf_output_1"]["id"]) + self._assert_has_keys(output_content , "id", "elements") + assert output_content["collection_type"] == "list:paired", output_content + elements = output_content["elements"] + assert len(elements) == 2 + elements0 = elements[0] + assert elements0["element_identifier"] == "el1" + + def test_workflow_run_input_mapping_with_subworkflows(self): + with self.dataset_populator.test_history() as history_id: + summary = self._run_jobs("""%s + +test_data: + outer_input: + type: list + name: the_dataset_list + elements: + - identifier: el1 + value: 1.fastq + type: File + - identifier: el2 + value: 1.fastq + type: File +""" % SIMPLE_NESTED_WORKFLOW_YAML, history_id=history_id) + workflow_id = summary.workflow_id + invocation_id = summary.invocation_id + invocation_response = self._get("workflows/%s/invocations/%s" % (workflow_id, invocation_id)) + self._assert_status_code_is(invocation_response, 200) + invocation_response = self._get("workflows/%s/invocations/%s" % (workflow_id, invocation_id)) + self._assert_status_code_is(invocation_response, 200) + invocation = invocation_response.json() + self._assert_has_keys(invocation , "id", "outputs", "output_collections") + assert len(invocation["output_collections"]) == 1, invocation + assert len(invocation["outputs"]) == 0 + output_content = self.dataset_populator.get_history_collection_details(history_id, content_id=invocation["output_collections"]["outer_output"]["id"]) + self._assert_has_keys(output_content , "id", "elements") + assert output_content["collection_type"] == "list", output_content + elements = output_content["elements"] + assert len(elements) == 2 + elements0 = elements[0] + assert elements0["element_identifier"] == "el1" + @skip_without_tool("cat") def test_cancel_new_workflow_when_history_deleted(self): with self.dataset_populator.test_history() as history_id: From 0185e04a5beaf951bb20d670d9421d988d70bb29 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Mon, 11 Sep 2017 14:28:26 -0400 Subject: [PATCH 09/22] Test recover_mapping in conjunction with subworkflow executions. --- test/api/test_workflows.py | 112 +++++++++++++++++++++++++++++++++++++ 1 file changed, 112 insertions(+) diff --git a/test/api/test_workflows.py b/test/api/test_workflows.py index b3501adc9a2..47d9820887a 100644 --- a/test/api/test_workflows.py +++ b/test/api/test_workflows.py @@ -1160,6 +1160,118 @@ test_data: elements0 = elements[0] assert elements0["element_identifier"] == "el1" + @skip_without_tool("cat_list") + @skip_without_tool("random_lines1") + @skip_without_tool("split") + def test_subworkflow_recover_mapping(self): + with self.dataset_populator.test_history() as history_id: + self._run_jobs(""" +class: GalaxyWorkflow +inputs: + - id: outer_input +outputs: + - id: outer_output + source: second_cat#out_file1 +steps: + - tool_id: cat1 + label: first_cat + state: + input1: + $link: outer_input + - run: + class: GalaxyWorkflow + inputs: + - id: inner_input + outputs: + - id: workflow_output + source: random_lines#out_file1 + steps: + - tool_id: random_lines1 + label: random_lines + state: + num_lines: 2 + input: + $link: inner_input + seed_source: + seed_source_selector: set_seed + seed: asdf + label: nested_workflow + connect: + inner_input: first_cat#out_file1 + - tool_id: split + label: split + state: + input1: + $link: nested_workflow#workflow_output + - tool_id: cat_list + label: second_cat + state: + input1: + $link: split#output + +test_data: + outer_input: + value: 1.bed + type: File +""", history_id=history_id, wait=True) + self.assertEqual("chr16\t142908\t143003\tCCDS10397.1_cds_0_0_chr16_142909_f\t0\t+\nchr5\t131424298\t131424460\tCCDS4149.1_cds_0_0_chr5_131424299_f\t0\t+\n", self.dataset_populator.get_history_dataset_content(history_id)) + + @skip_without_tool("cat_list") + @skip_without_tool("random_lines1") + @skip_without_tool("split") + def test_recover_mapping_in_subworkflow(self): + with self.dataset_populator.test_history() as history_id: + self._run_jobs(""" +class: GalaxyWorkflow +inputs: + - id: outer_input +outputs: + - id: outer_output + source: second_cat#out_file1 +steps: + - tool_id: cat1 + label: first_cat + state: + input1: + $link: outer_input + - run: + class: GalaxyWorkflow + inputs: + - id: inner_input + outputs: + - id: workflow_output + source: split#output + steps: + - tool_id: random_lines1 + label: random_lines + state: + num_lines: 2 + input: + $link: inner_input + seed_source: + seed_source_selector: set_seed + seed: asdf + - tool_id: split + label: split + state: + input1: + $link: random_lines#out_file1 + label: nested_workflow + connect: + inner_input: first_cat#out_file1 + - tool_id: cat_list + label: second_cat + state: + input1: + $link: nested_workflow#workflow_output + +test_data: + outer_input: + value: 1.bed + type: File +""", history_id=history_id, wait=True) + self.assertEqual("chr16\t142908\t143003\tCCDS10397.1_cds_0_0_chr16_142909_f\t0\t+\nchr5\t131424298\t131424460\tCCDS4149.1_cds_0_0_chr5_131424299_f\t0\t+\n", self.dataset_populator.get_history_dataset_content(history_id)) + @skip_without_tool("cat") def test_cancel_new_workflow_when_history_deleted(self): with self.dataset_populator.test_history() as history_id: From 9e471d3e97a6d6b6893ed243b6a48108076899c4 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Wed, 15 Nov 2017 08:54:23 -0500 Subject: [PATCH 10/22] Allow mapping over empty collections. If one maps a tool over an empty collection, one should get an empty output collection for each of the tool's outputs. This should have always been a valid operation I think, the code just wasn't quite structured right when the collection work was merged - there were too many places I was assuming the existence of at least one actual output or input that was mapped over to continue going. This reworks all of that and seems to handle empty collections appropriately. Fixes #4025 (not being informing the user of the operation, but actually by allowing the operation - which it should be allowed I think). --- lib/galaxy/dataset_collections/matching.py | 4 +- lib/galaxy/tools/__init__.py | 32 +++++--- lib/galaxy/tools/execute.py | 18 ++--- lib/galaxy/workflow/modules.py | 5 +- test/api/test_tools.py | 18 +++++ test/api/test_workflows.py | 80 +++++++++++++++++++ test/base/populators.py | 2 +- .../tools/for_workflows/count_list.xml | 16 ++++ .../tools/for_workflows/count_multi_file.xml | 16 ++++ .../tools/for_workflows/empty_list.xml | 22 +++++ test/functional/tools/samples_tool_conf.xml | 3 + 11 files changed, 191 insertions(+), 25 deletions(-) create mode 100644 test/functional/tools/for_workflows/count_list.xml create mode 100644 test/functional/tools/for_workflows/count_multi_file.xml create mode 100644 test/functional/tools/for_workflows/empty_list.xml diff --git a/lib/galaxy/dataset_collections/matching.py b/lib/galaxy/dataset_collections/matching.py index b8720ccf80b..94b4f2d5d66 100644 --- a/lib/galaxy/dataset_collections/matching.py +++ b/lib/galaxy/dataset_collections/matching.py @@ -64,7 +64,9 @@ class MatchingCollections(object): effective_structure = leaf for unlinked_structure in self.unlinked_structures: effective_structure = effective_structure.multiply(unlinked_structure) - linked_structure = self.linked_structure or leaf + linked_structure = self.linked_structure + if linked_structure is None: + linked_structure = leaf effective_structure = effective_structure.multiply(linked_structure) return None if effective_structure.is_leaf else effective_structure diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index f29e2dbf95f..9731214c194 100755 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -84,7 +84,10 @@ from galaxy.web import url_for from galaxy.web.form_builder import SelectField from galaxy.work.context import WorkRequestContext from tool_shed.util import common_util -from .execute import execute as execute_job +from .execute import ( + execute as execute_job, + MappingParameters, +) from .loader import ( imported_macro_paths, raw_tool_xml_tree, @@ -1244,8 +1247,6 @@ class Tool(object, Dictifiable): # Fixed set of input parameters may correspond to any number of jobs. # Expand these out to individual parameters for given jobs (tool executions). expanded_incomings, collection_info = expand_meta_parameters(trans, self, incoming) - if not expanded_incomings: - raise exceptions.MessageException('Tool execution failed, trying to run a tool over an empty collection.') # Remapping a single job to many jobs doesn't make sense, so disable # remap if multi-runs of tools are being used. @@ -1295,17 +1296,24 @@ class Tool(object, Dictifiable): err_data = {key: value for d in all_errors for (key, value) in d.items()} raise exceptions.MessageException(', '.join(msg for msg in err_data.values()), err_data=err_data) else: - execution_tracker = execute_job(trans, self, all_params, history=request_context.history, rerun_remap_job_id=rerun_remap_job_id, collection_info=collection_info) - if execution_tracker.successful_jobs: - return dict(out_data=execution_tracker.output_datasets, - num_jobs=len(execution_tracker.successful_jobs), - job_errors=execution_tracker.execution_errors, - jobs=execution_tracker.successful_jobs, - output_collections=execution_tracker.output_collections, - implicit_collections=execution_tracker.implicit_collections) - else: + mapping_params = MappingParameters(incoming, all_params) + execution_tracker = execute_job(trans, self, mapping_params, history=request_context.history, rerun_remap_job_id=rerun_remap_job_id, collection_info=collection_info) + # Raise an exception if there were jobs to execute and none of them were submitted, + # if at least one is submitted or there are no jobs to execute - return aggregate + # information including per-job errors. Arguably we should just always return the + # aggregate information - we just haven't done that historically. + raise_execution_exception = not execution_tracker.successful_jobs and len(all_params) > 0 + + if raise_execution_exception: raise exceptions.MessageException(execution_tracker.execution_errors[0]) + return dict(out_data=execution_tracker.output_datasets, + num_jobs=len(execution_tracker.successful_jobs), + job_errors=execution_tracker.execution_errors, + jobs=execution_tracker.successful_jobs, + output_collections=execution_tracker.output_collections, + implicit_collections=execution_tracker.implicit_collections) + def handle_single_execution(self, trans, rerun_remap_job_id, params, history, mapping_over_collection, execution_cache=None): """ Return a pair with whether execution is successful as well as either diff --git a/lib/galaxy/tools/execute.py b/lib/galaxy/tools/execute.py index e6d37538110..6e217e81215 100644 --- a/lib/galaxy/tools/execute.py +++ b/lib/galaxy/tools/execute.py @@ -24,12 +24,16 @@ class PartialJobExecution(Exception): self.jobs = jobs -def execute(trans, tool, param_combinations, history, rerun_remap_job_id=None, collection_info=None, workflow_invocation_uuid=None, invocation_step=None, max_num_jobs=None): +MappingParameters = collections.namedtuple("MappingParameters", ["param_template", "param_combinations"]) + + +def execute(trans, tool, mapping_params, history, rerun_remap_job_id=None, collection_info=None, workflow_invocation_uuid=None, invocation_step=None, max_num_jobs=None): """ Execute a tool and return object containing summary (output data, number of failures, etc...). """ all_jobs_timer = ExecutionTimer() + param_combinations = mapping_params.param_combinations execution_tracker = ToolExecutionTracker(tool, param_combinations, collection_info) app = trans.app execution_cache = ToolExecutionCache(trans) @@ -120,12 +124,8 @@ def execute(trans, tool, param_combinations, history, rerun_remap_job_id=None, c log.debug("Executed %d job(s) for tool %s request: %s" % (job_count, tool.id, all_jobs_timer)) if collection_info: history = history or tool.get_default_history_by_trans(trans) - if len(param_combinations) == 0: - template = "Attempting to map over an empty collection, this is not yet implemented. collection_info is [%s]" - message = template % collection_info - log.warning(message) - raise Exception(message) - params = param_combinations[0] + # TODO: this perhaps should always just be the param_template. Going to try to be safe first. -John + params = param_combinations[0] if param_combinations else mapping_params.param_template execution_tracker.create_output_collections(trans, history, params) return execution_tracker @@ -196,13 +196,13 @@ class ToolExecutionTracker(object): collections = {} implicit_inputs = list(self.collection_info.collections.items()) - for output_name, outputs in self.outputs_by_output_name.items(): + for output_name, output in self.tool.outputs.items(): + outputs = self.outputs_by_output_name[output_name] if not len(structure) == len(outputs): # Output does not have the same structure, if all jobs were # successfully submitted this shouldn't have happened. log.warning("Problem matching up datasets while attempting to create implicit dataset collections") continue - output = self.tool.outputs[output_name] element_identifiers = None if hasattr(output, "default_identifier_source"): diff --git a/lib/galaxy/workflow/modules.py b/lib/galaxy/workflow/modules.py index 7abc55636a5..0ad4609b915 100644 --- a/lib/galaxy/workflow/modules.py +++ b/lib/galaxy/workflow/modules.py @@ -21,7 +21,7 @@ from galaxy.tools import ( DefaultToolState, ToolInputsNotReadyException ) -from galaxy.tools.execute import execute, PartialJobExecution +from galaxy.tools.execute import execute, MappingParameters, PartialJobExecution from galaxy.tools.parameters import ( check_param, params_to_incoming, @@ -875,10 +875,11 @@ class ToolModule(WorkflowModule): # workflow should be delayed. partial_jobs = None try: + mapping_params = MappingParameters(tool_state.inputs, param_combinations) execution_tracker = execute( trans=self.trans, tool=tool, - param_combinations=param_combinations, + mapping_params=mapping_params, history=invocation.history, collection_info=collection_info, workflow_invocation_uuid=invocation.uuid.hex, diff --git a/test/api/test_tools.py b/test/api/test_tools.py index e11ce93c8be..c3e37ff74f3 100644 --- a/test/api/test_tools.py +++ b/test/api/test_tools.py @@ -681,6 +681,24 @@ class ToolsTestCase(api.ApiTestCase): } self._run_and_check_simple_collection_mapping(history_id, inputs) + @skip_without_tool("cat1") + def test_map_over_empty_collection(self): + with self.dataset_populator.test_history() as history_id: + hdca_id = self.dataset_collection_populator.create_list_in_history(history_id, contents=[]).json()['id'] + inputs = { + "input1": {'batch': True, 'values': [{'src': 'hdca', 'id': hdca_id}]}, + } + create = self._run_cat1(history_id, inputs=inputs, assert_ok=True) + outputs = create['outputs'] + jobs = create['jobs'] + implicit_collections = create['implicit_collections'] + self.assertEquals(len(jobs), 0) + self.assertEquals(len(outputs), 0) + self.assertEquals(len(implicit_collections), 1) + + empty_output = implicit_collections[0] + assert empty_output["name"] == "Concatenate datasets on collection 1", empty_output + @skip_without_tool("output_action_change_format") def test_map_over_with_output_format_actions(self): for use_action in ["do", "dont"]: diff --git a/test/api/test_workflows.py b/test/api/test_workflows.py index 47d9820887a..876fc167269 100644 --- a/test/api/test_workflows.py +++ b/test/api/test_workflows.py @@ -1272,6 +1272,86 @@ test_data: """, history_id=history_id, wait=True) self.assertEqual("chr16\t142908\t143003\tCCDS10397.1_cds_0_0_chr16_142909_f\t0\t+\nchr5\t131424298\t131424460\tCCDS4149.1_cds_0_0_chr5_131424299_f\t0\t+\n", self.dataset_populator.get_history_dataset_content(history_id)) + @skip_without_tool("empty_list") + @skip_without_tool("count_list") + @skip_without_tool("random_lines1") + def test_empty_list_mapping(self): + with self.dataset_populator.test_history() as history_id: + self._run_jobs(""" +class: GalaxyWorkflow +inputs: + - id: input1 +outputs: + - id: count_list + source: count_list#out_file1 +steps: + - tool_id: empty_list + label: empty_list + state: + input1: + $link: input1 + - tool_id: random_lines1 + label: random_lines + state: + num_lines: 2 + input: + $link: empty_list#output + seed_source: + seed_source_selector: set_seed + seed: asdf + - tool_id: count_list + label: count_list + state: + input1: + $link: random_lines#out_file1 + +test_data: + input1: + value: 1.bed + type: File +""", history_id=history_id, wait=True) + self.assertEqual("0\n", self.dataset_populator.get_history_dataset_content(history_id)) + + @skip_without_tool("empty_list") + @skip_without_tool("count_multi_file") + @skip_without_tool("random_lines1") + def test_empty_list_reduction(self): + with self.dataset_populator.test_history() as history_id: + self._run_jobs(""" +class: GalaxyWorkflow +inputs: + - id: input1 +outputs: + - id: count_multi_file + source: count_multi_file#out_file1 +steps: + - tool_id: empty_list + label: empty_list + state: + input1: + $link: input1 + - tool_id: random_lines1 + label: random_lines + state: + num_lines: 2 + input: + $link: empty_list#output + seed_source: + seed_source_selector: set_seed + seed: asdf + - tool_id: count_multi_file + label: count_multi_file + state: + input1: + $link: random_lines#out_file1 + +test_data: + input1: + value: 1.bed + type: File +""", history_id=history_id, wait=True) + self.assertEqual("0\n", self.dataset_populator.get_history_dataset_content(history_id)) + @skip_without_tool("cat") def test_cancel_new_workflow_when_history_deleted(self): with self.dataset_populator.test_history() as history_id: diff --git a/test/base/populators.py b/test/base/populators.py index b64b57ff526..45b3d1a13ec 100644 --- a/test/base/populators.py +++ b/test/base/populators.py @@ -615,7 +615,7 @@ class BaseDatasetCollectionPopulator(object): return element_identifiers def list_identifiers(self, history_id, contents=None): - count = 3 if not contents else len(contents) + count = 3 if contents is None else len(contents) # Contents can be a list of strings (with name auto-assigned here) or a list of # 2-tuples of form (name, dataset_content). if contents and isinstance(contents[0], tuple): diff --git a/test/functional/tools/for_workflows/count_list.xml b/test/functional/tools/for_workflows/count_list.xml new file mode 100644 index 00000000000..abb56f955c3 --- /dev/null +++ b/test/functional/tools/for_workflows/count_list.xml @@ -0,0 +1,16 @@ + + count the number of items in a list + '$out_file1' + ]]> + + + + + + + + + + + diff --git a/test/functional/tools/for_workflows/count_multi_file.xml b/test/functional/tools/for_workflows/count_multi_file.xml new file mode 100644 index 00000000000..c7a2c8ce124 --- /dev/null +++ b/test/functional/tools/for_workflows/count_multi_file.xml @@ -0,0 +1,16 @@ + + count the number of datasets in a multiple file input + '$out_file1' + ]]> + + + + + + + + + + + diff --git a/test/functional/tools/for_workflows/empty_list.xml b/test/functional/tools/for_workflows/empty_list.xml new file mode 100644 index 00000000000..df251d5db0e --- /dev/null +++ b/test/functional/tools/for_workflows/empty_list.xml @@ -0,0 +1,22 @@ + + always produce an empty list + + mkdir outputs; + cd outputs; + + + + + + + + + + + + + + + + + diff --git a/test/functional/tools/samples_tool_conf.xml b/test/functional/tools/samples_tool_conf.xml index 88ecc25f816..a469930b9c3 100644 --- a/test/functional/tools/samples_tool_conf.xml +++ b/test/functional/tools/samples_tool_conf.xml @@ -140,6 +140,9 @@ + + +
From 53cba4aa74b0d2543445824a7636f92249baa150 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Thu, 5 Oct 2017 14:39:53 -0400 Subject: [PATCH 11/22] Refactor collection manager into smaller, more focused methods. Stronger assertions to verify arguments are correct. --- lib/galaxy/managers/collections.py | 50 +++++++++++++++++++----------- 1 file changed, 32 insertions(+), 18 deletions(-) diff --git a/lib/galaxy/managers/collections.py b/lib/galaxy/managers/collections.py index 0e6420a43c3..c0e57fc9e51 100644 --- a/lib/galaxy/managers/collections.py +++ b/lib/galaxy/managers/collections.py @@ -105,37 +105,29 @@ class DatasetCollectionManager(object): message = "Internal logic error - create called with unknown parent type %s" % type(parent) log.exception(message) raise MessageException(message) - tags = tags or {} - if implicit_collection_info: - for _, v in implicit_collection_info.get('implicit_inputs', []): - for tag in [t for t in v.tags if t.user_tname == 'name']: - tags[tag.value] = tag - for _, tag in tags.items(): - dataset_collection_instance.tags.append(tag.copy(cls=model.HistoryDatasetCollectionTagAssociation)) + implicit_inputs = [] + if implicit_collection_info: + implicit_inputs = implicit_collection_info.get('implicit_inputs', []) + tags = self._append_tags(dataset_collection_instance, implicit_inputs, tags) return self.__persist(dataset_collection_instance) def create_dataset_collection(self, trans, collection_type, element_identifiers=None, elements=None, hide_source_items=None): + # Make sure at least one of these is None. + assert element_identifiers is None or elements is None + if element_identifiers is None and elements is None: raise RequestParameterInvalidException(ERROR_INVALID_ELEMENTS_SPECIFICATION) if not collection_type: raise RequestParameterInvalidException(ERROR_NO_COLLECTION_TYPE) + collection_type_description = self.collection_type_descriptions.for_collection_type(collection_type) + # If we have elements, this is an internal request, don't need to load # objects from identifiers. if elements is None: - if collection_type_description.has_subcollections(): - # Nested collection - recursively create collections and update identifiers. - self.__recursively_create_collections(trans, element_identifiers) - new_collection = False - for element_identifier in element_identifiers: - if element_identifier.get("src") == "new_collection" and element_identifier.get('collection_type') == '': - new_collection = True - elements = self.__load_elements(trans, element_identifier['element_identifiers']) - if not new_collection: - elements = self.__load_elements(trans, element_identifiers) - + elements = self._element_identifiers_to_elements(trans, collection_type_description, element_identifiers) # else if elements is set, it better be an ordered dict! if elements is not self.ELEMENTS_UNINITIALIZED: @@ -150,6 +142,28 @@ class DatasetCollectionManager(object): dataset_collection.collection_type = collection_type return dataset_collection + def _element_identifiers_to_elements(self, trans, collection_type_description, element_identifiers): + if collection_type_description.has_subcollections(): + # Nested collection - recursively create collections and update identifiers. + self.__recursively_create_collections(trans, element_identifiers) + new_collection = False + for element_identifier in element_identifiers: + if element_identifier.get("src") == "new_collection" and element_identifier.get('collection_type') == '': + new_collection = True + elements = self.__load_elements(trans, element_identifier['element_identifiers']) + if not new_collection: + elements = self.__load_elements(trans, element_identifiers) + return elements + + def _append_tags(self, dataset_collection_instance, implicit_inputs=None, tags=None): + tags = tags or {} + implicit_inputs = implicit_inputs or [] + for _, v in implicit_inputs: + for tag in [t for t in v.tags if t.user_tname == 'name']: + tags[tag.value] = tag + for _, tag in tags.items(): + dataset_collection_instance.tags.append(tag.copy(cls=model.HistoryDatasetCollectionTagAssociation)) + def set_collection_elements(self, dataset_collection, dataset_instances): if dataset_collection.populated: raise Exception("Cannot reset elements of an already populated dataset collection.") From 78babab6455f4cd1c51aed0b1ddf6b4b29ad6d03 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Wed, 4 Oct 2017 14:01:51 -0400 Subject: [PATCH 12/22] Pre-build collections during mapping to be more recoverable. Create mapped over collections ahead of time instead of at the end only if they are valid. - This way if there is a problem that problem can be reported in the state of the mapped collection. - This way datasets can just be hidden instead of displayed and then hidden (fixes #1790). - I think this should improve performance and prevent locks because we no longer have conflicting threads trying to write dataset state. - We populate the collection as we go so we can recover and continue scheduling the mapped step (needed for #3883) --- lib/galaxy/dataset_collections/matching.py | 8 +- lib/galaxy/dataset_collections/structure.py | 88 +++- .../dataset_collections/type_description.py | 1 + lib/galaxy/managers/collections.py | 65 ++- lib/galaxy/managers/collections_util.py | 16 +- lib/galaxy/model/__init__.py | 21 +- lib/galaxy/tools/__init__.py | 15 +- lib/galaxy/tools/actions/__init__.py | 21 +- lib/galaxy/tools/actions/model_operations.py | 4 +- lib/galaxy/tools/execute.py | 416 ++++++++++++------ lib/galaxy/tools/parser/output_objects.py | 5 +- lib/galaxy/workflow/modules.py | 32 +- lib/galaxy/workflow/run.py | 29 +- test/api/test_tools.py | 2 +- test/api/test_workflow_extraction.py | 2 +- test/api/test_workflows.py | 8 +- test/base/populators.py | 17 + .../unit/dataset_collections/test_matching.py | 1 + 18 files changed, 516 insertions(+), 235 deletions(-) diff --git a/lib/galaxy/dataset_collections/matching.py b/lib/galaxy/dataset_collections/matching.py index 94b4f2d5d66..aa62b626a9f 100644 --- a/lib/galaxy/dataset_collections/matching.py +++ b/lib/galaxy/dataset_collections/matching.py @@ -44,23 +44,29 @@ class MatchingCollections(object): self.linked_structure = None self.unlinked_structures = [] self.collections = {} + self.subcollection_types = {} def __attempt_add_to_linked_match(self, input_name, hdca, collection_type_description, subcollection_type): structure = get_structure(hdca, collection_type_description, leaf_subcollection_type=subcollection_type) if not self.linked_structure: self.linked_structure = structure self.collections[input_name] = hdca + self.subcollection_types[input_name] = subcollection_type else: if not self.linked_structure.can_match(structure): raise exceptions.MessageException(CANNOT_MATCH_ERROR_MESSAGE) self.collections[input_name] = hdca + self.subcollection_types[input_name] = subcollection_type def slice_collections(self): return self.linked_structure.walk_collections(self.collections) + def subcollection_mapping_type(self, input_name): + return self.subcollection_types[input_name] + @property def structure(self): - """Yield cross product of all unlinked datasets to linked dataset.""" + """Yield cross product of all unlinked collections structures to linked collection structure.""" effective_structure = leaf for unlinked_structure in self.unlinked_structures: effective_structure = effective_structure.multiply(unlinked_structure) diff --git a/lib/galaxy/dataset_collections/structure.py b/lib/galaxy/dataset_collections/structure.py index 3278dd65e40..19058b15b42 100644 --- a/lib/galaxy/dataset_collections/structure.py +++ b/lib/galaxy/dataset_collections/structure.py @@ -1,12 +1,17 @@ """ Module for reasoning about structure of and matching hierarchical collections of data. """ import logging -log = logging.getLogger(__name__) + +import six from .type_description import map_over_collection_type +log = logging.getLogger(__name__) + +@six.python_2_unicode_compatible class Leaf(object): + children_known = True def __len__(self): return 1 @@ -18,18 +23,60 @@ class Leaf(object): def clone(self): return self - def multiply(self, other_structure): - return other_structure.clone() + def multiply(self, other_structure, uninitialized=False): + if not uninitialized: + return other_structure.clone() + else: + return UnitializedTree(other_structure.collection_type_description) + + def sliced_collection_type(self, collection): + return input + + def __str__(self): + return "Leaf[]" leaf = Leaf() -class Tree(object): +class BaseTree(object): + + def __init__(self, collection_type_description): + self.collection_type_description = collection_type_description + + +@six.python_2_unicode_compatible +class UnitializedTree(BaseTree): + children_known = False + + def clone(self): + return self + + @property + def is_leaf(self): + return False + + def __len__(self): + raise Exception("Unknown length") + + def multiply(self, other_structure, uninitialized=False): + if other_structure.is_leaf: + return self.clone() + + new_collection_type = self.collection_type_description.multiply(other_structure.collection_type_description) + return UnitializedTree(new_collection_type) + + def __str__(self): + return "UnitializedTree[collection_type=%s]" % self.collection_type_description + + +@six.python_2_unicode_compatible +class Tree(BaseTree): + children_known = True def __init__(self, children, collection_type_description): + super(Tree, self).__init__(collection_type_description) self.children = children - self.collection_type_description = collection_type_description @staticmethod def for_dataset_collection(dataset_collection, collection_type_description): @@ -107,14 +154,14 @@ class Tree(object): element_identifiers=element_identifiers, ) - def multiply(self, other_structure): + def multiply(self, other_structure, uninitialized=False): if other_structure.is_leaf: return self.clone() new_collection_type = self.collection_type_description.multiply(other_structure.collection_type_description) new_children = [] for (identifier, structure) in self.children: - new_children.append((identifier, structure.multiply(other_structure))) + new_children.append((identifier, structure.multiply(other_structure, uninitialized=uninitialized))) return Tree(new_children, new_collection_type) @@ -122,6 +169,30 @@ class Tree(object): cloned_children = [(_[0], _[1].clone()) for _ in self.children] return Tree(cloned_children, self.collection_type_description) + def __str__(self): + return "Tree[collection_type=%s,children=%s]" % (self.collection_type_description, ",".join(map(lambda identifier_and_element: "%s=%s" % (identifier_and_element[0], identifier_and_element[1]), self.children))) + + +def tool_output_to_structure(get_sliced_input_collection_type, tool_output, collections_manager): + if not tool_output.collection: + tree = leaf + else: + collection_type_descriptions = collections_manager.collection_type_descriptions + # Okay this is ToolCollectionOutputStructure not a Structure - different + # concepts of structure. + if tool_output.dynamic_structure: + # Two cases collection_type_source and collection_type right? + tree = UnitializedTree(collection_type_descriptions.for_type_description("list")) # list is obviously wrong... + else: + structured_like = tool_output.structure.structured_like + if structured_like: + collection_type = get_sliced_input_collection_type(structured_like) + else: + collection_type = tool_output.structure.collection_type + tree = UnitializedTree(collection_type) + + return tree + def dict_map(func, input_dict): return dict((k, func(v)) for k, v in input_dict.items()) @@ -131,4 +202,5 @@ def get_structure(dataset_collection_instance, collection_type_description, leaf if leaf_subcollection_type: collection_type_description = collection_type_description.effective_collection_type_description(leaf_subcollection_type) - return Tree.for_dataset_collection(dataset_collection_instance.collection, collection_type_description) + collection = dataset_collection_instance.collection + return Tree.for_dataset_collection(collection, collection_type_description) diff --git a/lib/galaxy/dataset_collections/type_description.py b/lib/galaxy/dataset_collections/type_description.py index f7b8bed1eb7..ee76a4609b5 100644 --- a/lib/galaxy/dataset_collections/type_description.py +++ b/lib/galaxy/dataset_collections/type_description.py @@ -8,6 +8,7 @@ class CollectionTypeDescriptionFactory(object): self.type_registry = type_registry def for_collection_type(self, collection_type): + assert collection_type is not None return CollectionTypeDescription(collection_type, self) diff --git a/lib/galaxy/managers/collections.py b/lib/galaxy/managers/collections.py index c0e57fc9e51..8939eae80a8 100644 --- a/lib/galaxy/managers/collections.py +++ b/lib/galaxy/managers/collections.py @@ -46,6 +46,37 @@ class DatasetCollectionManager(object): self.tag_manager = tags.GalaxyTagManager(app.model.context) self.ldda_manager = lddas.LDDAManager(app) + def precreate_dataset_collection_instance(self, trans, parent, name, implicit_inputs, implicit_output_name, structure): + dataset_collection = self.precreate_dataset_collection(structure) + return self._create_instance_for_collection( + trans, parent, name, dataset_collection, implicit_inputs=implicit_inputs, implicit_output_name=implicit_output_name, + ) + + def precreate_dataset_collection(self, structure): + if structure.is_leaf or not structure.children_known: + return model.DatasetCollectionElement.UNINITIALIZED_ELEMENT + else: + collection_type_description = structure.collection_type_description + dataset_collection = model.DatasetCollection(populated=False) + dataset_collection.collection_type = collection_type_description.collection_type + elements = [] + for index, (identifier, substructure) in enumerate(structure.children): + # TODO: Open question - populate these now or later? + if substructure.is_leaf: + element = model.DatasetCollectionElement.UNINITIALIZED_ELEMENT + else: + element = self.precreate_dataset_collection(substructure) + + element = model.DatasetCollectionElement( + element=element, + element_identifier=identifier, + element_index=index, + ) + elements.append(element) + dataset_collection.elements = elements + + return dataset_collection + def create(self, trans, parent, name, collection_type, element_identifiers=None, elements=None, implicit_collection_info=None, trusted_identifiers=None, hide_source_items=False, tags=None): @@ -68,27 +99,30 @@ class DatasetCollectionManager(object): hide_source_items=hide_source_items, ) + implicit_inputs = [] + if implicit_collection_info: + implicit_inputs = implicit_collection_info.get('implicit_inputs', []) + + implicit_output_name = None + if implicit_collection_info: + implicit_output_name = implicit_collection_info["implicit_output_name"] + + return self._create_instance_for_collection( + trans, parent, name, dataset_collection, implicit_inputs=implicit_inputs, implicit_output_name=implicit_output_name, tags=tags + ) + + def _create_instance_for_collection(self, trans, parent, name, dataset_collection, implicit_output_name=None, implicit_inputs=None, tags=None): if isinstance(parent, model.History): dataset_collection_instance = self.model.HistoryDatasetCollectionAssociation( collection=dataset_collection, name=name, ) - if implicit_collection_info: - for input_name, input_collection in implicit_collection_info["implicit_inputs"]: + if implicit_inputs: + for input_name, input_collection in implicit_inputs: dataset_collection_instance.add_implicit_input_collection(input_name, input_collection) - for output_dataset in implicit_collection_info.get("outputs"): - if output_dataset not in trans.sa_session: - output_dataset = trans.sa_session.query(type(output_dataset)).get(output_dataset.id) - if isinstance(output_dataset, model.HistoryDatasetAssociation): - output_dataset.hidden_beneath_collection_instance = dataset_collection_instance - elif isinstance(output_dataset, model.HistoryDatasetCollectionAssociation): - dataset_collection_instance.add_implicit_input_collection(input_name, input_collection) - else: - # dataset collection, don't need to do anything... - pass - trans.sa_session.add(output_dataset) - dataset_collection_instance.implicit_output_name = implicit_collection_info["implicit_output_name"] + if implicit_output_name: + dataset_collection_instance.implicit_output_name = implicit_output_name log.debug("Created collection with %d elements" % (len(dataset_collection_instance.collection.elements))) # Handle setting hid @@ -106,9 +140,6 @@ class DatasetCollectionManager(object): log.exception(message) raise MessageException(message) - implicit_inputs = [] - if implicit_collection_info: - implicit_inputs = implicit_collection_info.get('implicit_inputs', []) tags = self._append_tags(dataset_collection_instance, implicit_inputs, tags) return self.__persist(dataset_collection_instance) diff --git a/lib/galaxy/managers/collections_util.py b/lib/galaxy/managers/collections_util.py index 831a7b8cdb7..190cbd2b929 100644 --- a/lib/galaxy/managers/collections_util.py +++ b/lib/galaxy/managers/collections_util.py @@ -122,12 +122,16 @@ def dictify_dataset_collection_instance(dataset_collection_instance, parent, sec def dictify_element(element): dictified = element.to_dict(view="element") - object_detials = element.element_object.to_dict() - if element.child_collection: - # Recursively yield elements for each nested collection... - child_collection = element.child_collection - object_detials["elements"] = [dictify_element(_) for _ in child_collection.elements] - object_detials["populated"] = child_collection.populated + element_object = element.element_object + if element_object is not None: + object_detials = element.element_object.to_dict() + if element.child_collection: + # Recursively yield elements for each nested collection... + child_collection = element.child_collection + object_detials["elements"] = [dictify_element(_) for _ in child_collection.elements] + object_detials["populated"] = child_collection.populated + else: + object_detials = None dictified["object"] = object_detials return dictified diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 707f96cbdbf..019dcae1530 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -3218,6 +3218,15 @@ class DatasetCollection(object, Dictifiable, UsesAnnotations): self.populated_state = DatasetCollection.populated_states.FAILED self.populated_state_message = message + def finalize(self): + # All jobs have written out their elements - everything should be populated + # but might not be - check that second case! (TODO) + self.mark_as_populated() + if self.has_subcollections: + # THIS IS WRONG - SHOULD ONLY BE TO THE DEPTH OF THE MAP OVER. + for element in self.elements: + element.child_collection.finalize() + @property def dataset_instances(self): instances = [] @@ -3485,6 +3494,8 @@ class DatasetCollectionElement(object, Dictifiable): dict_collection_visible_keys = ['id', 'element_type', 'element_index', 'element_identifier'] dict_element_visible_keys = ['id', 'element_type', 'element_index', 'element_identifier'] + UNINITIALIZED_ELEMENT = object() + def __init__( self, id=None, @@ -3499,7 +3510,7 @@ class DatasetCollectionElement(object, Dictifiable): self.ldda = element elif isinstance(element, DatasetCollection): self.child_collection = element - else: + elif element != self.UNINITIALIZED_ELEMENT: raise AttributeError('Unknown element type provided: %s' % type(element)) self.id = id @@ -3517,7 +3528,7 @@ class DatasetCollectionElement(object, Dictifiable): # TOOD: Rename element_type to element_type. return "dataset_collection" else: - raise Exception("Unknown element instance type") + return None @property def is_collection(self): @@ -3532,7 +3543,7 @@ class DatasetCollectionElement(object, Dictifiable): elif self.child_collection: return self.child_collection else: - raise Exception("Unknown element instance type") + return None @property def dataset_instance(self): @@ -4194,6 +4205,10 @@ class WorkflowInvocationStep(object, Dictifiable): def update(self): self.workflow_invocation.update() + @property + def is_new(self): + return self.state == self.states.NEW + def add_output(self, output_name, output_object): if output_object.history_content_type == "dataset": output_assoc = WorkflowInvocationStepOutputDatasetAssociation() diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index 9731214c194..4d872d4fc60 100755 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -1314,13 +1314,20 @@ class Tool(object, Dictifiable): output_collections=execution_tracker.output_collections, implicit_collections=execution_tracker.implicit_collections) - def handle_single_execution(self, trans, rerun_remap_job_id, params, history, mapping_over_collection, execution_cache=None): + def handle_single_execution(self, trans, rerun_remap_job_id, execution_slice, history, execution_cache=None): """ Return a pair with whether execution is successful as well as either resulting output data or an error message indicating the problem. """ try: - job, out_data = self.execute(trans, incoming=params, history=history, rerun_remap_job_id=rerun_remap_job_id, mapping_over_collection=mapping_over_collection, execution_cache=execution_cache) + job, out_data = self.execute( + trans, + incoming=execution_slice.param_combination, + history=history, + rerun_remap_job_id=rerun_remap_job_id, + execution_cache=execution_cache, + dataset_collection_elements=execution_slice.dataset_collection_elements, + ) except httpexceptions.HTTPFound as e: # if it's a paste redirect exception, pass it up the stack raise e @@ -2263,7 +2270,7 @@ class DatabaseOperationTool(Tool): def check_inputs_ready(self, input_datasets, input_dataset_collections): def check_dataset_instance(input_dataset): if input_dataset.is_pending: - raise ToolInputsNotReadyException() + raise ToolInputsNotReadyException("An input dataset is pending.") if self.require_dataset_ok: if input_dataset.state != input_dataset.dataset.states.OK: @@ -2275,7 +2282,7 @@ class DatabaseOperationTool(Tool): for input_dataset_collection_pairs in input_dataset_collections.values(): for input_dataset_collection, is_mapped in input_dataset_collection_pairs: if not input_dataset_collection.collection.populated: - raise ToolInputsNotReadyException() + raise ToolInputsNotReadyException("An input collection is not populated.") map(check_dataset_instance, input_dataset_collection.dataset_instances) diff --git a/lib/galaxy/tools/actions/__init__.py b/lib/galaxy/tools/actions/__init__.py index 66bbac2c78a..14bd8f54452 100644 --- a/lib/galaxy/tools/actions/__init__.py +++ b/lib/galaxy/tools/actions/__init__.py @@ -195,7 +195,7 @@ class DefaultToolAction(object): return history, inp_data, inp_dataset_collections - def execute(self, tool, trans, incoming={}, return_job=False, set_output_hid=True, history=None, job_params=None, rerun_remap_job_id=None, mapping_over_collection=False, execution_cache=None): + def execute(self, tool, trans, incoming={}, return_job=False, set_output_hid=True, history=None, job_params=None, rerun_remap_job_id=None, execution_cache=None, dataset_collection_elements=None): """ Executes a tool, creating job and tool outputs, associating them, and submitting the job to the job queue. If history is not specified, use @@ -267,7 +267,7 @@ class DefaultToolAction(object): tool=tool, tool_action=self, input_collections=input_collections, - mapping_over_collection=mapping_over_collection, + dataset_collection_elements=dataset_collection_elements, on_text=on_text, incoming=incoming, params=wrapped_params.params, @@ -304,8 +304,12 @@ class DefaultToolAction(object): data = app.model.HistoryDatasetAssociation(extension=ext, create_dataset=True, flush=False) if hidden is None: hidden = output.hidden + if not hidden and dataset_collection_elements is not None: # Mapping over a collection - hide datasets + hidden = True if hidden: data.visible = False + if dataset_collection_elements is not None and name in dataset_collection_elements: + dataset_collection_elements[name].hda = data trans.sa_session.add(data) trans.app.security_agent.set_all_dataset_permissions(data.dataset, output_permissions, new=True) for _, tag in preserved_tags.items(): @@ -351,6 +355,7 @@ class DefaultToolAction(object): for name, output in tool.outputs.items(): if not filter_output(output, incoming): + handle_output_timer = ExecutionTimer() if output.collection: collections_manager = app.dataset_collections_service element_identifiers = [] @@ -402,15 +407,14 @@ class DefaultToolAction(object): element_kwds = dict(elements=collections_manager.ELEMENTS_UNINITIALIZED) else: element_kwds = dict(element_identifiers=element_identifiers) - output_collections.create_collection( output=output, name=name, tags=preserved_tags, **element_kwds ) + log.info("Handled collection output named %s for tool %s %s" % (name, tool.id, handle_output_timer)) else: - handle_output_timer = ExecutionTimer() handle_output(name, output) log.info("Handled output named %s for tool %s %s" % (name, tool.id, handle_output_timer)) @@ -670,13 +674,13 @@ class OutputCollections(object): parameter). """ - def __init__(self, trans, history, tool, tool_action, input_collections, mapping_over_collection, on_text, incoming, params, job_params): + def __init__(self, trans, history, tool, tool_action, input_collections, dataset_collection_elements, on_text, incoming, params, job_params): self.trans = trans self.history = history self.tool = tool self.tool_action = tool_action self.input_collections = input_collections - self.mapping_over_collection = mapping_over_collection + self.dataset_collection_elements = dataset_collection_elements self.on_text = on_text self.incoming = incoming self.params = params @@ -710,12 +714,15 @@ class OutputCollections(object): for dataset in value.dataset_instances: assert dataset.history is not None - if self.mapping_over_collection: + if self.dataset_collection_elements is not None: dc = collections_manager.create_dataset_collection( self.trans, collection_type=collection_type, **element_kwds ) + if name in self.dataset_collection_elements: + self.dataset_collection_elements[name].child_collection = dc + # self.trans.sa_session.add(self.dataset_collection_elements[name]) self.out_collections[name] = dc else: hdca_name = self.tool_action.get_output_name( diff --git a/lib/galaxy/tools/actions/model_operations.py b/lib/galaxy/tools/actions/model_operations.py index 4edcd6ce141..842a3c085d5 100644 --- a/lib/galaxy/tools/actions/model_operations.py +++ b/lib/galaxy/tools/actions/model_operations.py @@ -21,7 +21,7 @@ class ModelOperationToolAction(DefaultToolAction): tool.check_inputs_ready(inp_data, inp_dataset_collections) - def execute(self, tool, trans, incoming={}, set_output_hid=False, overwrite=True, history=None, job_params=None, mapping_over_collection=False, execution_cache=None, **kwargs): + def execute(self, tool, trans, incoming={}, set_output_hid=False, overwrite=True, history=None, job_params=None, execution_cache=None, **kwargs): if execution_cache is None: execution_cache = ToolExecutionCache(trans) @@ -42,7 +42,7 @@ class ModelOperationToolAction(DefaultToolAction): tool=tool, tool_action=self, input_collections=input_collections, - mapping_over_collection=mapping_over_collection, + dataset_collection_elements=kwargs.get("dataset_collection_elements", None), on_text=on_text, incoming=incoming, params=wrapped_params.params, diff --git a/lib/galaxy/tools/execute.py b/lib/galaxy/tools/execute.py index 6e217e81215..34794d6d338 100644 --- a/lib/galaxy/tools/execute.py +++ b/lib/galaxy/tools/execute.py @@ -4,12 +4,15 @@ from various states, tracking results, and building implicit dataset collections from matched collections. """ import collections +import itertools import logging from threading import Thread from six.moves.queue import Queue -from galaxy.tools.actions import on_text_for_names, ToolExecutionCache +from galaxy import model +from galaxy.dataset_collections.structure import tool_output_to_structure +from galaxy.tools.actions import filter_output, on_text_for_names, ToolExecutionCache from galaxy.tools.parser import ToolOutputCollectionPart from galaxy.util import ExecutionTimer @@ -20,47 +23,49 @@ EXECUTION_SUCCESS_MESSAGE = "Tool [%s] created job [%s] %s" class PartialJobExecution(Exception): - def __init__(self, jobs): - self.jobs = jobs + def __init__(self): + pass MappingParameters = collections.namedtuple("MappingParameters", ["param_template", "param_combinations"]) -def execute(trans, tool, mapping_params, history, rerun_remap_job_id=None, collection_info=None, workflow_invocation_uuid=None, invocation_step=None, max_num_jobs=None): +def execute(trans, tool, mapping_params, history, rerun_remap_job_id=None, collection_info=None, workflow_invocation_uuid=None, invocation_step=None, max_num_jobs=None, job_callback=None): """ Execute a tool and return object containing summary (output data, number of failures, etc...). """ + if max_num_jobs: + assert invocation_step is not None + if rerun_remap_job_id: + assert invocation_step is None + all_jobs_timer = ExecutionTimer() - param_combinations = mapping_params.param_combinations - execution_tracker = ToolExecutionTracker(tool, param_combinations, collection_info) + if invocation_step is None: + execution_tracker = ToolExecutionTracker(tool, mapping_params, collection_info) + else: + execution_tracker = WorkflowStepExecutionTracker(tool, mapping_params, collection_info, invocation_step, job_callback=job_callback) app = trans.app execution_cache = ToolExecutionCache(trans) - new_jobs = [] - - def execute_single_job(params): + def execute_single_job(execution_slice): job_timer = ExecutionTimer() + params = execution_slice.param_combination if workflow_invocation_uuid: params['__workflow_invocation_uuid__'] = workflow_invocation_uuid elif '__workflow_invocation_uuid__' in params: # Only workflow invocation code gets to set this, ignore user supplied # values or rerun parameters. del params['__workflow_invocation_uuid__'] - job, result = tool.handle_single_execution(trans, rerun_remap_job_id, params, history, collection_info, execution_cache) + + job, result = tool.handle_single_execution(trans, rerun_remap_job_id, execution_slice, history, execution_cache) if job: message = EXECUTION_SUCCESS_MESSAGE % (tool.id, job.id, job_timer) log.debug(message) - execution_tracker.record_success(job, result) - new_jobs.append(job) + execution_tracker.record_success(execution_slice, job, result) else: execution_tracker.record_error(result) - config = app.config - burst_at = getattr(config, 'tool_submission_burst_at', 10) - burst_threads = getattr(config, 'tool_submission_burst_threads', 1) - tool_action = tool.tool_action if hasattr(tool_action, "check_inputs_ready"): for params in execution_tracker.param_combinations: @@ -72,24 +77,23 @@ def execute(trans, tool, mapping_params, history, rerun_remap_job_id=None, colle history ) - if invocation_step: - execution_tracker.recover_successful_jobs(invocation_step) + execution_tracker.ensure_implicit_collections_populated(trans, history, mapping_params.param_template) + config = app.config + burst_at = getattr(config, 'tool_submission_burst_at', 10) + burst_threads = getattr(config, 'tool_submission_burst_threads', 1) - previously_executed_jobs_count = len(execution_tracker.successful_jobs) job_count = len(execution_tracker.param_combinations) jobs_executed = 0 has_remaining_jobs = False if (job_count < burst_at or burst_threads < 2): - for index, params in enumerate(execution_tracker.param_combinations): - if index < previously_executed_jobs_count: - continue - elif max_num_jobs and jobs_executed >= max_num_jobs: + for execution_slice in execution_tracker.new_execution_slices(): + if max_num_jobs and jobs_executed >= max_num_jobs: has_remaining_jobs = True break else: - execute_single_job(params) + execute_single_job(execution_slice) jobs_executed += 1 else: # TODO: re-record success... @@ -106,59 +110,240 @@ def execute(trans, tool, mapping_params, history, rerun_remap_job_id=None, colle t.daemon = True t.start() - for index, params in enumerate(execution_tracker.param_combinations): - if index < previously_executed_jobs_count: - continue - elif max_num_jobs and jobs_executed >= max_num_jobs: + for execution_slice in execution_tracker.new_execution_slices(): + if max_num_jobs and jobs_executed >= max_num_jobs: has_remaining_jobs = True break else: - q.put(params) + q.put(execution_slice) jobs_executed += 1 q.join() if has_remaining_jobs: - raise PartialJobExecution(new_jobs) + raise PartialJobExecution() + else: + execution_tracker.finalize_dataset_collections(trans) log.debug("Executed %d job(s) for tool %s request: %s" % (job_count, tool.id, all_jobs_timer)) - if collection_info: - history = history or tool.get_default_history_by_trans(trans) - # TODO: this perhaps should always just be the param_template. Going to try to be safe first. -John - params = param_combinations[0] if param_combinations else mapping_params.param_template - execution_tracker.create_output_collections(trans, history, params) - return execution_tracker -class ToolExecutionTracker(object): +class ExecutionSlice(object): - def __init__(self, tool, param_combinations, collection_info): + def __init__(self, job_index, param_combination, dataset_collection_elements=None): + self.job_index = job_index + self.param_combination = param_combination + self.dataset_collection_elements = dataset_collection_elements + + +class ExecutionTracker(object): + + def __init__(self, tool, mapping_params, collection_info): + # Known ahead of time... self.tool = tool - self.param_combinations = param_combinations + self.mapping_params = mapping_params self.collection_info = collection_info - self.successful_jobs = [] + + self._on_text = None + + # Populated as we go... self.failed_jobs = 0 self.execution_errors = [] + + self.successful_jobs = [] self.output_datasets = [] self.output_collections = [] - self.outputs_by_output_name = collections.defaultdict(list) + self.implicit_collections = {} - def recover_successful_jobs(self, invocation_step): - # TODO: Optimize away the need to do this - we should just be dealing with IDs - # and such and we shouldn't fetch them until the very end when we need them to create - # collections. - for job_assoc in invocation_step.jobs: - job = job_assoc.job - for job_output in job.output_datasets: - self.outputs_by_output_name[job_output.name].append(job_output.dataset) - for job_output in job.output_dataset_collections: - self.outputs_by_output_name[job_output.name].append(job_output.dataset_collection) + @property + def param_combinations(self): + return self.mapping_params.param_combinations - def record_success(self, job, outputs): + @property + def example_params(self): + if self.mapping_params.param_combinations: + return self.mapping_params.param_combinations[0] + else: + # TODO: This isn't quite right - what we want is something like param_template wrapped, + # need a test case with an output filter applied to an empty list, still this is + # an improvement over not allowing mapping of empty lists. + return self.mapping_params.param_template + + @property + def job_count(self): + return len(self.param_combinations) + + def record_error(self, error): + self.failed_jobs += 1 + message = "There was a failure executing a job for tool [%s] - %s" + log.warning(message, self.tool.id, error) + self.execution_errors.append(error) + + @property + def on_text(self): + if self._on_text is None: + collection_names = ["collection %d" % c.hid for c in self.collection_info.collections.values()] + self._on_text = on_text_for_names(collection_names) + + return self._on_text + + def output_name(self, trans, history, params, output): + on_text = self.on_text + + try: + output_collection_name = self.tool.tool_action.get_output_name( + output, + dataset=None, + tool=self.tool, + on_text=on_text, + trans=trans, + history=history, + params=params, + incoming=None, + job_params=None, + ) + except Exception: + output_collection_name = "%s across %s" % (self.tool.name, on_text) + + return output_collection_name + + def sliced_input_collection_type(self, input_name): + if self.is_implicit_input(input_name): + subcollection_mapping_type = self.collection_info.subcollection_mapping_type(input_name) + return subcollection_mapping_type + # return self.collection_info.structure.sliced_input_collection_type(self.implicit_inputs[input_name]) + else: + return self.mapping_params.param_template[input_name].collection.collection_type + + def _structure_for_output(self, trans, tool_output): + structure = self.collection_info.structure + if hasattr(tool_output, "default_identifier_source"): + # Switch the structure for outputs if the output specified a default_identifier_source + collection_type_descriptions = trans.app.dataset_collections_service.collection_type_descriptions + + source_collection = self.collection_info.collections.get(tool_output.default_identifier_source) + if source_collection: + collection_type_description = collection_type_descriptions.for_collection_type(source_collection.collection.collection_type) + _structure = structure.for_dataset_collection(source_collection.collection, collection_type_description=collection_type_description) + if structure.can_match(_structure): + structure = _structure + + return structure + + def _element_identifiers_for_output(self, trans, tool_output, outputs): + output_structure = self._structure_for_output(trans, tool_output) + element_identifiers = output_structure.element_identifiers_for_outputs(trans, outputs) + return element_identifiers + + def _mapped_output_structure(self, trans, tool_output): + collections_manager = trans.app.dataset_collections_service + output_structure = tool_output_to_structure(self.sliced_input_collection_type, tool_output, collections_manager) + mapping_structure = self._structure_for_output(trans, tool_output) + # Output structure may not be known, but input structure must be, + # otherwise this step of the workflow shouldn't have been scheduled + # or the tool should not have been executable on this input. + mapped_output_structure = mapping_structure.multiply(output_structure, uninitialized=True) + return mapped_output_structure + + def ensure_implicit_collections_populated(self, trans, history, params): + if not self.collection_info: + return + + history = history or self.tool.get_default_history_by_trans(trans) + # params = param_combinations[0] if param_combinations else mapping_params.param_template + self.precreate_output_collections(trans, history, params) + + def precreate_output_collections(self, trans, history, params): + # params is just one sample tool param execution with parallelized + # collection replaced with a specific dataset. Need to replace this + # with the collection and wrap everything up so can evaluate output + # label. + params.update(self.collection_info.collections) # Replace datasets with source collections for labelling outputs. + + collection_instances = {} + implicit_inputs = self.implicit_inputs + for output_name, output in self.tool.outputs.items(): + if filter_output(output, self.example_params): + continue + output_collection_name = self.output_name(trans, history, params, output) + effective_structure = self._mapped_output_structure(trans, output) + collection_instance = trans.app.dataset_collections_service.precreate_dataset_collection_instance( + trans=trans, + parent=history, + name=output_collection_name, + implicit_inputs=implicit_inputs, + implicit_output_name=output_name, + structure=effective_structure, + ) + collection_instances[output_name] = collection_instance + trans.sa_session.add(collection_instance) + # Needed to flush the association created just above with + # job.add_output_dataset_collection. + trans.sa_session.flush() + self.implicit_collections = collection_instances + + def finalize_dataset_collections(self, trans): + # TODO: this probably needs to be reworked some, we should have the collection methods + # return a list of changed objects to add to the session and flush and we should only + # be finalizing collections to a depth of self.collection_info.structure. So for instance + # if you are mapping a list over a tool that dynamically generates lists - we won't actually + # know the structure of the inner list until after its job is complete. + if self.failed_jobs > 0: + for implicit_collection in self.implicit_collections.values(): + implicit_collection.collection.handle_population_failed("One or more jobs failed during dataset initialization.") + trans.sa_session.add(implicit_collection.collection) + else: + for implicit_collection in self.implicit_collections.values(): + implicit_collection.collection.finalize() + trans.sa_session.add(implicit_collection.collection) + trans.sa_session.flush() + + @property + def implicit_inputs(self): + implicit_inputs = list(self.collection_info.collections.items()) + return implicit_inputs + + def is_implicit_input(self, input_name): + return input_name in self.collection_info.collections + + def walk_implicit_collections(self): + return self.collection_info.structure.walk_collections(self.implicit_collections) + + def new_execution_slices(self): + if self.collection_info is None: + for job_index, param_combination in enumerate(self.param_combinations): + yield ExecutionSlice(job_index, param_combination) + else: + for execution_slice in self.new_collection_execution_slices(): + yield execution_slice + + def record_success(self, execution_slice, job, outputs): + # TODO: successful_jobs need to be inserted in the correct place... self.successful_jobs.append(job) self.output_datasets.extend(outputs) + for job_output in job.output_dataset_collection_instances: + self.output_collections.append((job_output.name, job_output.dataset_collection_instance)) + if self.implicit_collections: + for output_name, collection_instance in self.implicit_collections.items(): + job.add_output_dataset_collection(output_name, collection_instance) + + +# Seperate these because workflows need to track their jobs belong to the invocation +# in the database immediately and they can be recovered. +class ToolExecutionTracker(ExecutionTracker): + + def __init__(self, tool, mapping_params, collection_info): + super(ToolExecutionTracker, self).__init__(tool, mapping_params, collection_info) + + # New to track these things for tool output API response in the tool case, + # in the workflow case we just write stuff to the database and forget about + # it. + self.outputs_by_output_name = collections.defaultdict(list) + + def record_success(self, execution_slice, job, outputs): + super(ToolExecutionTracker, self).record_success(execution_slice, job, outputs) for output_name, output_dataset in outputs: if ToolOutputCollectionPart.is_named_collection_part_name(output_name): # Skip known collection outputs, these will be covered by @@ -167,101 +352,58 @@ class ToolExecutionTracker(object): self.outputs_by_output_name[output_name].append(output_dataset) for job_output in job.output_dataset_collections: self.outputs_by_output_name[job_output.name].append(job_output.dataset_collection) - for job_output in job.output_dataset_collection_instances: - self.output_collections.append((job_output.name, job_output.dataset_collection_instance)) - def record_error(self, error): - self.failed_jobs += 1 - message = "There was a failure executing a job for tool [%s] - %s" - log.warning(message, self.tool.id, error) - self.execution_errors.append(error) + def new_collection_execution_slices(self): + for job_index, (param_combination, dataset_collection_elements) in enumerate(itertools.izip(self.param_combinations, self.walk_implicit_collections())): + for dataset_collection_element in dataset_collection_elements.values(): + assert dataset_collection_element.element_object is None - def create_output_collections(self, trans, history, params): - # TODO: Move this function - it doesn't belong here but it does need - # the information in this class and potential extensions. - if self.failed_jobs > 0: - return [] + yield ExecutionSlice(job_index, param_combination, dataset_collection_elements) - structure = self.collection_info.structure - # params is just one sample tool param execution with parallelized - # collection replaced with a specific dataset. Need to replace this - # with the collection and wrap everything up so can evaluate output - # label. - params.update(self.collection_info.collections) # Replace datasets with source collections for labelling outputs. +class WorkflowStepExecutionTracker(ExecutionTracker): - collection_names = ["collection %d" % c.hid for c in self.collection_info.collections.values()] - on_text = on_text_for_names(collection_names) + def __init__(self, tool, mapping_params, collection_info, invocation_step, job_callback): + super(WorkflowStepExecutionTracker, self).__init__(tool, mapping_params, collection_info) + self.invocation_step = invocation_step + self.job_callback = job_callback - collections = {} + def record_success(self, execution_slice, job, outputs): + super(WorkflowStepExecutionTracker, self).record_success(execution_slice, job, outputs) + job_assoc = model.WorkflowInvocationStepJobAssociation() + job_assoc.index = execution_slice.job_index + job_assoc.workflow_invocation_step = self.invocation_step + job_assoc.job_id = job.id + self.job_callback(job) - implicit_inputs = list(self.collection_info.collections.items()) - for output_name, output in self.tool.outputs.items(): - outputs = self.outputs_by_output_name[output_name] - if not len(structure) == len(outputs): - # Output does not have the same structure, if all jobs were - # successfully submitted this shouldn't have happened. - log.warning("Problem matching up datasets while attempting to create implicit dataset collections") + def new_collection_execution_slices(self): + for job_index, (param_combination, dataset_collection_elements) in enumerate(itertools.izip(self.param_combinations, self.walk_implicit_collections())): + # Two options here - check if the element has been populated or check if the + # a WorkflowInvocationStepJobAssociation exists. Not sure which is better but + # for now I have the first so lets check. + found_result = False + for dataset_collection_element in dataset_collection_elements.values(): + if dataset_collection_element.element_object is not None: + found_result = True + break + if found_result: continue + yield ExecutionSlice(job_index, param_combination, dataset_collection_elements) - element_identifiers = None - if hasattr(output, "default_identifier_source"): - # Switch the structure for outputs if the output specified a default_identifier_source - collection_type_descriptions = trans.app.dataset_collections_service.collection_type_descriptions + def ensure_implicit_collections_populated(self, trans, history, params): + if not self. collection_info: + return - source_collection = self.collection_info.collections.get(output.default_identifier_source) - if source_collection: - collection_type_description = collection_type_descriptions.for_collection_type(source_collection.collection.collection_type) - _structure = structure.for_dataset_collection(source_collection.collection, collection_type_description=collection_type_description) - if structure.can_match(_structure): - element_identifiers = _structure.element_identifiers_for_outputs(trans, outputs) - - if not element_identifiers: - element_identifiers = structure.element_identifiers_for_outputs(trans, outputs) - - implicit_collection_info = dict( - implicit_inputs=implicit_inputs, - implicit_output_name=output_name, - outputs=outputs - ) - try: - output_collection_name = self.tool.tool_action.get_output_name( - output, - dataset=None, - tool=self.tool, - on_text=on_text, - trans=trans, - history=history, - params=params, - incoming=None, - job_params=None, - ) - except Exception: - output_collection_name = "%s across %s" % (self.tool.name, on_text) - - child_element_identifiers = element_identifiers["element_identifiers"] - collection_type = element_identifiers["collection_type"] - collection = trans.app.dataset_collections_service.create( - trans=trans, - parent=history, - name=output_collection_name, - element_identifiers=child_element_identifiers, - collection_type=collection_type, - implicit_collection_info=implicit_collection_info, - ) - for job in self.successful_jobs: - # TODO: Think through this, may only want this for output - # collections - or we may be already recording data in some - # other way. - if job not in trans.sa_session: - job = trans.sa_session.query(trans.app.model.Job).get(job.id) - job.add_output_dataset_collection(output_name, collection) - collections[output_name] = collection - - # Needed to flush the association created just above with - # job.add_output_dataset_collection. - trans.sa_session.flush() - self.implicit_collections = collections + history = history or self.tool.get_default_history_by_trans(trans) + if self.invocation_step.is_new: + self.precreate_output_collections(trans, history, params) + else: + collections = {} + for output_assoc in self.invocation_step.output_dataset_collections: + implicit_collection = output_assoc.dataset_collection + assert hasattr(implicit_collection, "history_content_type") # make sure it is an HDCA and not a DC + collections[output_assoc.output_name] = output_assoc.dataset_collection + self.implicit_collections = collections __all__ = ('execute', ) diff --git a/lib/galaxy/tools/parser/output_objects.py b/lib/galaxy/tools/parser/output_objects.py index 3dd543a3744..e680c6717be 100644 --- a/lib/galaxy/tools/parser/output_objects.py +++ b/lib/galaxy/tools/parser/output_objects.py @@ -195,7 +195,10 @@ class ToolOutputCollectionStructure(object): if self.structured_like: collection_prototype = inputs[self.structured_like].collection else: - collection_prototype = type_registry.prototype(self.collection_type) + collection_type = self.collection_type + assert collection_type + collection_prototype = type_registry.prototype(collection_type) + collection_prototype.collection_type = collection_type return collection_prototype diff --git a/lib/galaxy/workflow/modules.py b/lib/galaxy/workflow/modules.py index 0ad4609b915..ba44feeb5e6 100644 --- a/lib/galaxy/workflow/modules.py +++ b/lib/galaxy/workflow/modules.py @@ -220,8 +220,9 @@ class WorkflowModule(object): Use the supplied workflow progress object to track outputs, find inputs, etc.... - Return jobs created and a boolean indicating if there are additional - jobs to create on subsequent workflow scheduling tests. + Return a False if there is additional processing required to + on subsequent workflow scheduling runs, None or True means the workflow + step executed properly. """ raise TypeError("Abstract method") @@ -871,9 +872,7 @@ class ToolModule(WorkflowModule): param_combinations.append(execution_state.inputs) - # Will be set if only a subset of required jobs have been scheduled and the - # workflow should be delayed. - partial_jobs = None + complete = False try: mapping_params = MappingParameters(tool_state.inputs, param_combinations) execution_tracker = execute( @@ -884,36 +883,29 @@ class ToolModule(WorkflowModule): collection_info=collection_info, workflow_invocation_uuid=invocation.uuid.hex, invocation_step=invocation_step, + job_callback=lambda job: self._handle_post_job_actions(step, job, invocation.replacement_dict), ) - except PartialJobExecution as p: - partial_jobs = p.jobs + complete = True + except PartialJobExecution: + pass + except ToolInputsNotReadyException: delayed_why = "tool [%s] inputs are not ready, this special tool requires inputs to be ready" % tool.id raise DelayedWorkflowEvaluation(why=delayed_why) - if partial_jobs is None: + if invocation_step.is_new: if collection_info: step_outputs = dict(execution_tracker.implicit_collections) else: step_outputs = dict(execution_tracker.output_datasets) step_outputs.update(execution_tracker.output_collections) progress.set_step_outputs(invocation_step, step_outputs) - jobs = execution_tracker.successful_jobs - else: - jobs = partial_jobs - - for job in jobs: - self._handle_post_job_actions(step, job, invocation.replacement_dict) if execution_tracker.execution_errors: - failed_count = len(execution_tracker.execution_errors) - success_count = len(execution_tracker.successful_jobs) - all_count = failed_count + success_count - message = "Failed to create %d out of %s job(s) for workflow step." % (failed_count, all_count) + message = "Failed to create one or more job(s) for workflow step." raise Exception(message) - complete = partial_jobs is None - return jobs, complete + return complete def recover_mapping(self, invocation_step, progress): outputs = {} diff --git a/lib/galaxy/workflow/run.py b/lib/galaxy/workflow/run.py index 75222d5a46d..9a3c409243c 100644 --- a/lib/galaxy/workflow/run.py +++ b/lib/galaxy/workflow/run.py @@ -1,7 +1,7 @@ import logging import uuid -from galaxy import model, util +from galaxy import model from galaxy.util import ExecutionTimer from galaxy.util.odict import odict from galaxy.workflow import modules @@ -165,7 +165,6 @@ class WorkflowInvoker(object): for (step, workflow_invocation_step) in remaining_steps: step_delayed = False step_timer = ExecutionTimer() - jobs = None try: self.__check_implicitly_dependent_steps(step) @@ -174,29 +173,11 @@ class WorkflowInvoker(object): workflow_invocation_step.workflow_invocation = workflow_invocation workflow_invocation_step.workflow_step = step workflow_invocation_step.state = 'new' - previously_executed_jobs_count = 0 workflow_invocation.steps.append(workflow_invocation_step) - else: - previously_executed_jobs_count = len(workflow_invocation_step.jobs) - jobs_or_none = self._invoke_step(workflow_invocation_step) - if jobs_or_none: - jobs, complete = jobs_or_none - else: - jobs, complete = [], True - - for job in (util.listify(jobs) or []): - job_assoc = model.WorkflowInvocationStepJobAssociation() - job_assoc.index = previously_executed_jobs_count - job_assoc.workflow_invocation_step = workflow_invocation_step - # Job may not be generated in this thread if bursting is enabled - # https://github.com/galaxyproject/galaxy/issues/2259 - job_assoc.job_id = job.id - - previously_executed_jobs_count += 1 - - if not complete: + incomplete_or_none = self._invoke_step(workflow_invocation_step) + if incomplete_or_none is False: step_delayed = delayed_steps = True workflow_invocation_step.state = 'ready' self.progress.mark_step_outputs_delayed(step, why="Not all jobs scheduled for state.") @@ -268,8 +249,8 @@ class WorkflowInvoker(object): pass def _invoke_step(self, invocation_step): - jobs_or_none = invocation_step.workflow_step.module.execute(self.trans, self.progress, invocation_step) - return jobs_or_none + incomplete_or_none = invocation_step.workflow_step.module.execute(self.trans, self.progress, invocation_step) + return incomplete_or_none STEP_OUTPUT_DELAYED = object() diff --git a/test/api/test_tools.py b/test/api/test_tools.py index c3e37ff74f3..49a819909d6 100644 --- a/test/api/test_tools.py +++ b/test/api/test_tools.py @@ -1001,7 +1001,7 @@ class ToolsTestCase(api.ApiTestCase): self.assertEquals(len(jobs), 2) self.assertEquals(len(implicit_collections), 1) implicit_collection = implicit_collections[0] - assert implicit_collection["collection_type"] == "list:paired", implicit_collection + assert implicit_collection["collection_type"] == "list:paired", implicit_collection["collection_type"] outer_elements = implicit_collection["elements"] assert len(outer_elements) == 2 diff --git a/test/api/test_workflow_extraction.py b/test/api/test_workflow_extraction.py index 57008511576..d2cc98378c8 100644 --- a/test/api/test_workflow_extraction.py +++ b/test/api/test_workflow_extraction.py @@ -467,7 +467,7 @@ test_data: disconnected_inputs.append(value) if disconnected_inputs: - template = "%d step(s_ disconnected in extracted workflow - disconnectect steps are %s - workflow is %s" + template = "%d steps disconnected in extracted workflow - disconnectect steps are %s - workflow is %s" message = template % (len(disconnected_inputs), disconnected_inputs, workflow) raise AssertionError(message) diff --git a/test/api/test_workflows.py b/test/api/test_workflows.py index 876fc167269..ff03293ebb5 100644 --- a/test/api/test_workflows.py +++ b/test/api/test_workflows.py @@ -768,7 +768,9 @@ steps: } invocation_id = self.__invoke_workflow(history_id, workflow_id, inputs) self.wait_for_invocation_and_jobs(history_id, workflow_id, invocation_id) - content = self.dataset_populator.get_history_dataset_content(history_id, hid=7) + collection_details = self.dataset_populator.get_history_collection_details(history_id, hid=7) + assert collection_details["populated_state"] == "ok" + content = self.dataset_populator.get_history_dataset_content(history_id, hid=11) self.assertEqual(content.strip(), "samp1\t10.0\nsamp2\t20.0") @skip_without_tool("collection_split_on_column") @@ -1621,7 +1623,7 @@ test_data: def wait_for_invocation_and_jobs(self, history_id, workflow_id, invocation_id, assert_ok=True): self.workflow_populator.wait_for_invocation(workflow_id, invocation_id) time.sleep(.5) - self.dataset_populator.wait_for_history(history_id, assert_ok=assert_ok) + self.dataset_populator.wait_for_history_jobs(history_id, assert_ok=assert_ok) time.sleep(.5) def test_cannot_run_inaccessible_workflow(self): @@ -1740,7 +1742,7 @@ test_data: value: 1.fastq type: File """, history_id=history_id) - content = self.dataset_populator.get_history_dataset_details(history_id, hid=3, wait=True, assert_ok=True) + content = self.dataset_populator.get_history_dataset_details(history_id, hid=4, wait=True, assert_ok=True) name = content["name"] assert name == "my new name", name diff --git a/test/base/populators.py b/test/base/populators.py index 45b3d1a13ec..ae98d60d15e 100644 --- a/test/base/populators.py +++ b/test/base/populators.py @@ -163,6 +163,23 @@ class BaseDatasetPopulator(object): self._summarize_history(history_id) raise + def wait_for_history_jobs(self, history_id, assert_ok=False, timeout=DEFAULT_TIMEOUT): + query_params = {"history_id": history_id} + + def has_active_jobs(): + jobs_response = self._get("jobs", query_params) + assert jobs_response.status_code == 200 + active_jobs = [j for j in jobs_response.json() if j["state"] in ["new", "upload", "waiting", "queued", "running"]] + + if len(active_jobs) == 0: + return True + else: + return None + + wait_on(has_active_jobs, "active jobs", timeout=timeout) + if assert_ok: + return self.wait_for_history(history_id, assert_ok=True, timeout=timeout) + def wait_for_job(self, job_id, assert_ok=False, timeout=DEFAULT_TIMEOUT): return wait_on_state(lambda: self.get_job_details(job_id), assert_ok=assert_ok, timeout=timeout) diff --git a/test/unit/dataset_collections/test_matching.py b/test/unit/dataset_collections/test_matching.py index c27056abeb3..b3aa022479d 100644 --- a/test/unit/dataset_collections/test_matching.py +++ b/test/unit/dataset_collections/test_matching.py @@ -117,6 +117,7 @@ class MockCollection(object): def __init__(self, collection_type, elements): self.collection_type = collection_type self.elements = elements + self.populated = True class MockCollectionElement(object): From 40fd72a1980bd2df5ea116aae1016c8bad6c3ca1 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Mon, 16 Oct 2017 22:15:33 -0400 Subject: [PATCH 13/22] Allow rescheduling of single workflow steps. Implement maximum_workflow_jobs_per_scheduling_iteration to test this and mitigate memory issues associated with scheduling large workflow steps (maybe). Fixes #3883. --- config/galaxy.ini.sample | 8 ++ lib/galaxy/config.py | 1 + lib/galaxy/tools/execute.py | 6 +- lib/galaxy/workflow/modules.py | 20 ++-- lib/galaxy/workflow/run.py | 25 +++-- ...st_maximum_worklfow_invocation_duration.py | 48 ---------- .../test_workflow_scheduling_options.py | 92 +++++++++++++++++++ 7 files changed, 134 insertions(+), 66 deletions(-) delete mode 100644 test/integration/test_maximum_worklfow_invocation_duration.py create mode 100644 test/integration/test_workflow_scheduling_options.py diff --git a/config/galaxy.ini.sample b/config/galaxy.ini.sample index 11d457d892c..1d14bbb025a 100644 --- a/config/galaxy.ini.sample +++ b/config/galaxy.ini.sample @@ -1131,6 +1131,14 @@ use_interactive = True # invocation to schedule indefinitely. The default corresponds to 1 month. #maximum_workflow_invocation_duration = 2678400 +# Specify a maximum number of jobs that any given workflow scheduling iteration can create. +# Set this to a positive integer to prevent large collection jobs in a workflow from +# preventing other jobs from executing. This may also mitigate memory issues associated with +# scheduling workflows at the expense of increased total DB traffic because model objects +# are expunged from the SQL alchemy session between workflow invocation scheduling iterations. +# Set to -1 to disable any such maximum (the default). +#maximum_workflow_jobs_per_scheduling_iteration = -1 + # Force serial scheduling of workflows within the context of a particular history #history_local_serial_workflow_scheduling=False diff --git a/lib/galaxy/config.py b/lib/galaxy/config.py index af2a9aa0142..72fbd663208 100644 --- a/lib/galaxy/config.py +++ b/lib/galaxy/config.py @@ -388,6 +388,7 @@ class Configuration(object): self.history_local_serial_workflow_scheduling = string_as_bool(kwargs.get('history_local_serial_workflow_scheduling', 'False')) self.parallelize_workflow_scheduling_within_histories = string_as_bool(kwargs.get('parallelize_workflow_scheduling_within_histories', 'False')) self.maximum_workflow_invocation_duration = int(kwargs.get("maximum_workflow_invocation_duration", 2678400)) + self.maximum_workflow_jobs_per_scheduling_iteration = int(kwargs.get("maximum_workflow_jobs_per_scheduling_iteration", -1)) self.cache_user_job_count = string_as_bool(kwargs.get('cache_user_job_count', False)) self.pbs_application_server = kwargs.get('pbs_application_server', "") diff --git a/lib/galaxy/tools/execute.py b/lib/galaxy/tools/execute.py index 34794d6d338..7e96ce8c914 100644 --- a/lib/galaxy/tools/execute.py +++ b/lib/galaxy/tools/execute.py @@ -23,8 +23,8 @@ EXECUTION_SUCCESS_MESSAGE = "Tool [%s] created job [%s] %s" class PartialJobExecution(Exception): - def __init__(self): - pass + def __init__(self, execution_tracker): + self.execution_tracker = execution_tracker MappingParameters = collections.namedtuple("MappingParameters", ["param_template", "param_combinations"]) @@ -121,7 +121,7 @@ def execute(trans, tool, mapping_params, history, rerun_remap_job_id=None, colle q.join() if has_remaining_jobs: - raise PartialJobExecution() + raise PartialJobExecution(execution_tracker) else: execution_tracker.finalize_dataset_collections(trans) diff --git a/lib/galaxy/workflow/modules.py b/lib/galaxy/workflow/modules.py index ba44feeb5e6..80cec0cbd97 100644 --- a/lib/galaxy/workflow/modules.py +++ b/lib/galaxy/workflow/modules.py @@ -875,6 +875,7 @@ class ToolModule(WorkflowModule): complete = False try: mapping_params = MappingParameters(tool_state.inputs, param_combinations) + max_num_jobs = progress.maximum_jobs_to_schedule_or_none execution_tracker = execute( trans=self.trans, tool=tool, @@ -883,23 +884,24 @@ class ToolModule(WorkflowModule): collection_info=collection_info, workflow_invocation_uuid=invocation.uuid.hex, invocation_step=invocation_step, + max_num_jobs=max_num_jobs, job_callback=lambda job: self._handle_post_job_actions(step, job, invocation.replacement_dict), ) complete = True - except PartialJobExecution: - pass + except PartialJobExecution as pje: + execution_tracker = pje.execution_tracker except ToolInputsNotReadyException: delayed_why = "tool [%s] inputs are not ready, this special tool requires inputs to be ready" % tool.id raise DelayedWorkflowEvaluation(why=delayed_why) - if invocation_step.is_new: - if collection_info: - step_outputs = dict(execution_tracker.implicit_collections) - else: - step_outputs = dict(execution_tracker.output_datasets) - step_outputs.update(execution_tracker.output_collections) - progress.set_step_outputs(invocation_step, step_outputs) + progress.record_executed_job_count(len(execution_tracker.successful_jobs)) + if collection_info: + step_outputs = dict(execution_tracker.implicit_collections) + else: + step_outputs = dict(execution_tracker.output_datasets) + step_outputs.update(execution_tracker.output_collections) + progress.set_step_outputs(invocation_step, step_outputs, already_persisted=not invocation_step.is_new) if execution_tracker.execution_errors: message = "Failed to create one or more job(s) for workflow step." diff --git a/lib/galaxy/workflow/run.py b/lib/galaxy/workflow/run.py index 9a3c409243c..a1c76b2bf80 100644 --- a/lib/galaxy/workflow/run.py +++ b/lib/galaxy/workflow/run.py @@ -140,12 +140,18 @@ class WorkflowInvoker(object): module_injector = modules.WorkflowModuleInjector(trans) if progress is None: - progress = WorkflowProgress(self.workflow_invocation, workflow_run_config.inputs, module_injector) + progress = WorkflowProgress( + self.workflow_invocation, + workflow_run_config.inputs, + module_injector, + jobs_per_scheduling_iteration=getattr(trans.app.config, "maximum_workflow_jobs_per_scheduling_iteration", -1), + ) self.progress = progress def invoke(self): workflow_invocation = self.workflow_invocation - maximum_duration = getattr(self.trans.app.config, "maximum_workflow_invocation_duration", -1) + config = self.trans.app.config + maximum_duration = getattr(config, "maximum_workflow_invocation_duration", -1) if maximum_duration > 0 and workflow_invocation.seconds_since_created > maximum_duration: log.debug("Workflow invocation [%s] exceeded maximum number of seconds allowed for scheduling [%s], failing." % (workflow_invocation.id, maximum_duration)) workflow_invocation.state = model.WorkflowInvocation.states.FAILED @@ -258,16 +264,23 @@ STEP_OUTPUT_DELAYED = object() class WorkflowProgress(object): - def __init__(self, workflow_invocation, inputs_by_step_id, module_injector): + def __init__(self, workflow_invocation, inputs_by_step_id, module_injector, jobs_per_scheduling_iteration=-1): self.outputs = odict() self.module_injector = module_injector self.workflow_invocation = workflow_invocation self.inputs_by_step_id = inputs_by_step_id - self.jobs_per_scheduling_iteration = 1 + self.jobs_per_scheduling_iteration = jobs_per_scheduling_iteration + self.jobs_scheduled_this_iteration = 0 @property - def maximum_jobs_to_schedule(self): - return 1 + def maximum_jobs_to_schedule_or_none(self): + if self.jobs_per_scheduling_iteration > 0: + return self.jobs_per_scheduling_iteration - self.jobs_scheduled_this_iteration + else: + return None + + def record_executed_job_count(self, job_count): + self.jobs_scheduled_this_iteration += job_count def remaining_steps(self): # Previously computed and persisted step states. diff --git a/test/integration/test_maximum_worklfow_invocation_duration.py b/test/integration/test_maximum_worklfow_invocation_duration.py deleted file mode 100644 index b06661b4448..00000000000 --- a/test/integration/test_maximum_worklfow_invocation_duration.py +++ /dev/null @@ -1,48 +0,0 @@ -"""Integration tests for maximum workflow invocation duration configuration option.""" - -import time - -from json import dumps - -from base import integration_util -from base.populators import ( - DatasetPopulator, - WorkflowPopulator, -) - - -class MaximumWorkflowInvocationDurationTestCase(integration_util.IntegrationTestCase): - """Start a Pulsar job.""" - - framework_tool_and_types = True - - def setUp(self): - super(MaximumWorkflowInvocationDurationTestCase, self).setUp() - self.dataset_populator = DatasetPopulator(self.galaxy_interactor) - self.workflow_populator = WorkflowPopulator(self.galaxy_interactor) - - @classmethod - def handle_galaxy_config_kwds(cls, config): - config["maximum_workflow_invocation_duration"] = 20 - - def do_test(self): - workflow = self.workflow_populator.load_workflow_from_resource("test_workflow_pause") - workflow_id = self.workflow_populator.create_workflow(workflow) - history_id = self.dataset_populator.new_history() - hda1 = self.dataset_populator.new_dataset(history_id, content="1 2 3") - index_map = { - '0': dict(src="hda", id=hda1["id"]) - } - request = {} - request["history"] = "hist_id=%s" % history_id - request["inputs"] = dumps(index_map) - request["inputs_by"] = 'step_index' - url = "workflows/%s/invocations" % (workflow_id) - invocation_response = self._post(url, data=request) - invocation_url = url + "/" + invocation_response.json()["id"] - time.sleep(5) - state = self._get(invocation_url).json()["state"] - assert state != "failed", state - time.sleep(35) - state = self._get(invocation_url).json()["state"] - assert state == "failed", state diff --git a/test/integration/test_workflow_scheduling_options.py b/test/integration/test_workflow_scheduling_options.py new file mode 100644 index 00000000000..e1df0ba3417 --- /dev/null +++ b/test/integration/test_workflow_scheduling_options.py @@ -0,0 +1,92 @@ +"""Integration tests for workflow scheduling configuration option.""" + +import time + +from json import dumps + +from base import integration_util +from base.populators import ( + DatasetCollectionPopulator, + DatasetPopulator, + WorkflowPopulator, +) + + +class MaximumWorkflowInvocationDurationTestCase(integration_util.IntegrationTestCase): + + framework_tool_and_types = True + + def setUp(self): + super(MaximumWorkflowInvocationDurationTestCase, self).setUp() + self.dataset_populator = DatasetPopulator(self.galaxy_interactor) + self.workflow_populator = WorkflowPopulator(self.galaxy_interactor) + + @classmethod + def handle_galaxy_config_kwds(cls, config): + config["maximum_workflow_invocation_duration"] = 20 + + def do_test(self): + workflow = self.workflow_populator.load_workflow_from_resource("test_workflow_pause") + workflow_id = self.workflow_populator.create_workflow(workflow) + history_id = self.dataset_populator.new_history() + hda1 = self.dataset_populator.new_dataset(history_id, content="1 2 3") + index_map = { + '0': dict(src="hda", id=hda1["id"]) + } + request = {} + request["history"] = "hist_id=%s" % history_id + request["inputs"] = dumps(index_map) + request["inputs_by"] = 'step_index' + url = "workflows/%s/invocations" % (workflow_id) + invocation_response = self._post(url, data=request) + invocation_url = url + "/" + invocation_response.json()["id"] + time.sleep(5) + state = self._get(invocation_url).json()["state"] + assert state != "failed", state + time.sleep(35) + state = self._get(invocation_url).json()["state"] + assert state == "failed", state + + +class MaximumWorkflowJobsPerSchedulingIterationTestCase(integration_util.IntegrationTestCase): + + framework_tool_and_types = True + + def setUp(self): + super(MaximumWorkflowJobsPerSchedulingIterationTestCase, self).setUp() + self.dataset_populator = DatasetPopulator(self.galaxy_interactor) + self.workflow_populator = WorkflowPopulator(self.galaxy_interactor) + self.dataset_collection_populator = DatasetCollectionPopulator(self.galaxy_interactor) + + @classmethod + def handle_galaxy_config_kwds(cls, config): + config["maximum_workflow_jobs_per_scheduling_iteration"] = 1 + + def do_test(self): + workflow_id = self.workflow_populator.upload_yaml_workflow(""" +class: GalaxyWorkflow +steps: + - type: input_collection + - tool_id: collection_creates_pair + state: + input1: + $link: 0 + - tool_id: collection_paired_test + state: + f1: + $link: 1#paired_output + - tool_id: cat_list + state: + input1: + $link: 2#out1 +""") + with self.dataset_populator.test_history() as history_id: + hdca1 = self.dataset_collection_populator.create_list_in_history(history_id, contents=["a\nb\nc\nd\n", "e\nf\ng\nh\n"]).json() + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + inputs = { + '0': {"src": "hdca", "id": hdca1["id"]}, + } + invocation_id = self.workflow_populator.invoke_workflow(history_id, workflow_id, inputs) + self.workflow_populator.wait_for_workflow(history_id, workflow_id, invocation_id) + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + self.assertEqual("a\nc\nb\nd\ne\ng\nf\nh\n", self.dataset_populator.get_history_dataset_content(history_id, hid=0)) From d69962e7f4de38963401a032ada5131e6190b5e5 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Wed, 15 Nov 2017 08:56:10 -0500 Subject: [PATCH 14/22] Implement model abstraction for ``ImplicitCollectionJobs``. These aggregate all the jobs contributing to an implicitly created HistoryDatasetCollectionAssociation. - Model enhancements that allow this. - API endpoint for jobs summary of history content (somewhat uniform across HDAs and HDCAs just for consistency). - Test cases. - Rework history contents API a bit to eliminate more costly state summary checks by default. --- lib/galaxy/managers/collections.py | 15 +++-- lib/galaxy/managers/hdcas.py | 21 +++--- lib/galaxy/managers/jobs.py | 50 +++++++++++++- lib/galaxy/model/__init__.py | 43 ++++++++++++ lib/galaxy/model/mapping.py | 54 ++++++++++++++- .../versions/0136_record_workflow_outputs.py | 43 +++++++++++- lib/galaxy/tools/actions/__init__.py | 1 + lib/galaxy/tools/execute.py | 66 ++++++++++++------- .../webapps/galaxy/api/history_contents.py | 34 ++++++++++ lib/galaxy/webapps/galaxy/buildapp.py | 4 ++ scripts/db_shell.py | 1 + test/api/test_history_contents.py | 37 ++++++++++- test/api/test_tools.py | 23 +++---- test/base/populators.py | 7 ++ 14 files changed, 343 insertions(+), 56 deletions(-) diff --git a/lib/galaxy/managers/collections.py b/lib/galaxy/managers/collections.py index 8939eae80a8..356729593a9 100644 --- a/lib/galaxy/managers/collections.py +++ b/lib/galaxy/managers/collections.py @@ -47,10 +47,12 @@ class DatasetCollectionManager(object): self.ldda_manager = lddas.LDDAManager(app) def precreate_dataset_collection_instance(self, trans, parent, name, implicit_inputs, implicit_output_name, structure): + # TODO: prebuild all required HIDs and send them in so no need to flush in between. dataset_collection = self.precreate_dataset_collection(structure) - return self._create_instance_for_collection( - trans, parent, name, dataset_collection, implicit_inputs=implicit_inputs, implicit_output_name=implicit_output_name, + instance = self._create_instance_for_collection( + trans, parent, name, dataset_collection, implicit_inputs=implicit_inputs, implicit_output_name=implicit_output_name, flush=False ) + return instance def precreate_dataset_collection(self, structure): if structure.is_leaf or not structure.children_known: @@ -111,7 +113,7 @@ class DatasetCollectionManager(object): trans, parent, name, dataset_collection, implicit_inputs=implicit_inputs, implicit_output_name=implicit_output_name, tags=tags ) - def _create_instance_for_collection(self, trans, parent, name, dataset_collection, implicit_output_name=None, implicit_inputs=None, tags=None): + def _create_instance_for_collection(self, trans, parent, name, dataset_collection, implicit_output_name=None, implicit_inputs=None, tags=None, flush=True): if isinstance(parent, model.History): dataset_collection_instance = self.model.HistoryDatasetCollectionAssociation( collection=dataset_collection, @@ -141,7 +143,7 @@ class DatasetCollectionManager(object): raise MessageException(message) tags = self._append_tags(dataset_collection_instance, implicit_inputs, tags) - return self.__persist(dataset_collection_instance) + return self.__persist(dataset_collection_instance, flush=flush) def create_dataset_collection(self, trans, collection_type, element_identifiers=None, elements=None, hide_source_items=None): @@ -285,10 +287,11 @@ class DatasetCollectionManager(object): collections = list(filter(query.direct_match, collections)) return collections - def __persist(self, dataset_collection_instance): + def __persist(self, dataset_collection_instance, flush=True): context = self.model.context context.add(dataset_collection_instance) - context.flush() + if flush: + context.flush() return dataset_collection_instance def __recursively_create_collections(self, trans, element_identifiers): diff --git a/lib/galaxy/managers/hdcas.py b/lib/galaxy/managers/hdcas.py index e6591c61f4d..877c27390ab 100644 --- a/lib/galaxy/managers/hdcas.py +++ b/lib/galaxy/managers/hdcas.py @@ -117,12 +117,12 @@ class DCSerializer(base.ModelSerializer): 'create_time', 'update_time', 'collection_type', - 'populated', 'populated_state', 'populated_state_message', ]) self.add_view('detailed', [ - 'elements' + 'populated', + 'elements', ], include_keys_from='summary') def add_serializers(self): @@ -130,7 +130,7 @@ class DCSerializer(base.ModelSerializer): self.serializers.update({ 'model_class' : lambda *a, **c: 'DatasetCollection', 'elements' : self.serialize_elements, - 'element_count' : self.serialize_element_count + 'element_count' : self.serialize_element_count, }) def serialize_elements(self, item, key, **context): @@ -163,12 +163,12 @@ class DCASerializer(base.ModelSerializer): 'id', 'create_time', 'update_time', 'collection_type', - 'populated', 'populated_state', 'populated_state_message', ]) self.add_view('detailed', [ - 'elements' + 'populated', + 'elements', ], include_keys_from='summary') def add_serializers(self): @@ -184,7 +184,7 @@ class DCASerializer(base.ModelSerializer): 'populated_state', 'populated_state_message', 'elements', - 'element_count' + 'element_count', ] for key in collection_keys: self.serializers[key] = self._proxy_to_dataset_collection(key=key) @@ -221,15 +221,14 @@ class HDCASerializer( 'history_content_type', 'collection_type', - 'populated', 'populated_state', 'populated_state_message', + 'job_source_id', + 'job_source_type', + 'name', 'type_id', - 'history_id', - 'hid', - 'history_content_type', 'deleted', # 'purged', 'visible', @@ -238,6 +237,7 @@ class HDCASerializer( 'tags', # TODO: detail view only (maybe) ]) self.add_view('detailed', [ + 'populated', 'elements' ], include_keys_from='summary') @@ -254,6 +254,7 @@ class HDCASerializer( 'history_id' : self.serialize_id, 'history_content_type' : lambda *a, **c: self.hdca_manager.model_class.content_type, 'type_id' : self.serialize_type_id, + 'job_source_id' : self.serialize_id, 'url' : lambda i, k, **c: self.url_for('history_content_typed', history_id=self.app.security.encode_id(i.history_id), diff --git a/lib/galaxy/managers/jobs.py b/lib/galaxy/managers/jobs.py index a291d377b23..b967f897c9e 100644 --- a/lib/galaxy/managers/jobs.py +++ b/lib/galaxy/managers/jobs.py @@ -3,8 +3,9 @@ import logging from boltons.iterutils import remap from six import string_types -from sqlalchemy import and_, false, or_ +from sqlalchemy import and_, false, func, or_ from sqlalchemy.orm import aliased +from sqlalchemy.sql import select from galaxy import model from galaxy.managers.collections import DatasetCollectionManager @@ -243,3 +244,50 @@ class JobSearch(object): log.info("Searching jobs finished %s", search_timer) return job return None + + +def summarize_jobs_to_dict(sa_session, jobs_source): + """Proudce a summary of jobs for job summary endpoints. + + :type jobs_source: a Job or ImplicitCollectionJobs or None + :param jobs_source: the object to summarize + + :rtype: dict + :returns: dictionary containing job summary information + """ + rval = None + if jobs_source is None: + pass + elif isinstance(jobs_source, model.Job): + rval = { + "populated_state": "ok", + "states": {jobs_source.state: 1}, + "model": "Job", + "id": jobs_source.id, + } + else: + populated_state = jobs_source.populated_state + rval = { + "id": jobs_source.id, + "populated_state": populated_state, + "model": "ImplicitCollectionJobs", + } + if populated_state == "ok": + # produce state summary... + states = {} + join = model.ImplicitCollectionJobs.table.join( + model.ImplicitCollectionJobsJobAssociation.table.join(model.Job) + ) + statement = select( + [model.Job.state, func.count("*")] + ).select_from( + join + ).where( + model.ImplicitCollectionJobs.id == jobs_source.id + ).group_by( + model.Job.state + ) + for row in sa_session.execute(statement): + states[row[0]] = row[1] + rval["states"] = states + return rval diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 019dcae1530..a495bb65a75 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -1054,6 +1054,29 @@ class ImplicitlyCreatedDatasetCollectionInput(object): self.input_dataset_collection = input_dataset_collection +class ImplicitCollectionJobs(object): + + populated_states = Bunch( + NEW='new', # New implicit jobs object, unpopulated job associations + OK='ok', # Job associations are set and fixed. + FAILED='failed', # There were issues populating job associations, object is in error. + ) + + def __init__( + self, + id=None, + populated_state=None, + ): + self.id = id + self.populated_state = populated_state or ImplicitCollectionJobs.populated_states.NEW + + +class ImplicitCollectionJobsJobAssociation(object): + + def __init__(self): + pass + + class PostJobAction(object): def __init__(self, action_type, workflow_step, output_name=None, action_arguments=None): self.action_type = action_type @@ -3402,6 +3425,19 @@ class HistoryDatasetCollectionAssociation(DatasetCollectionInstance, return ((type_coerce(cls.content_type, types.Unicode) + u'-' + type_coerce(cls.id, types.Unicode)).label('type_id')) + @property + def job_source_type(self): + if self.implicit_collection_jobs_id: + return "ImplicitCollectionJobs" + elif self.job_id: + return "Job" + else: + return None + + @property + def job_source_id(self): + return self.implicit_collection_jobs_id or self.job_id + def to_hda_representative(self, multiple=False): rval = [] for dataset in self.collection.dataset_elements: @@ -3419,6 +3455,8 @@ class HistoryDatasetCollectionAssociation(DatasetCollectionInstance, history_content_type=self.history_content_type, visible=self.visible, deleted=self.deleted, + job_source_id=self.job_source_id, + job_source_type=self.job_source_type, **self._base_to_dict(view=view) ) @@ -3450,6 +3488,11 @@ class HistoryDatasetCollectionAssociation(DatasetCollectionInstance, name=self.name, copied_from_history_dataset_collection_association=self, ) + if self.implicit_collection_jobs_id: + hdca.implicit_collection_jobs_id = self.implicit_collection_jobs_id + elif self.job_id: + hdca.job_id = self.job_id + collection_copy = self.collection.copy( destination=hdca, element_destination=element_destination, diff --git a/lib/galaxy/model/mapping.py b/lib/galaxy/model/mapping.py index 49d6d4ed206..f53799db844 100644 --- a/lib/galaxy/model/mapping.py +++ b/lib/galaxy/model/mapping.py @@ -557,6 +557,20 @@ model.ImplicitlyCreatedDatasetCollectionInput.table = Table( ForeignKey("history_dataset_collection_association.id"), index=True), Column("name", Unicode(255))) +model.ImplicitCollectionJobs.table = Table( + "implicit_collection_jobs", metadata, + Column("id", Integer, primary_key=True), + Column("populated_state", TrimmedString(64), default='new', nullable=False), +) + +model.ImplicitCollectionJobsJobAssociation.table = Table( + "implicit_collection_jobs_job_association", metadata, + Column("id", Integer, primary_key=True), + Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), index=True), + Column("job_id", Integer, ForeignKey("job.id"), index=True), # Consider making this nullable... + Column("order_index", Integer, nullable=False), +) + model.JobExternalOutputMetadata.table = Table( "job_external_output_metadata", metadata, Column("id", Integer, primary_key=True), @@ -716,7 +730,10 @@ model.HistoryDatasetCollectionAssociation.table = Table( Column("deleted", Boolean, default=False), Column("copied_from_history_dataset_collection_association_id", Integer, ForeignKey("history_dataset_collection_association.id"), nullable=True), - Column("implicit_output_name", Unicode(255), nullable=True)) + Column("implicit_output_name", Unicode(255), nullable=True), + Column("job_id", ForeignKey("job.id"), index=True, nullable=True), + Column("implicit_collection_jobs_id", ForeignKey("implicit_collection_jobs.id"), index=True, nullable=True), +) model.LibraryDatasetCollectionAssociation.table = Table( "library_dataset_collection_association", metadata, @@ -2111,6 +2128,31 @@ simple_mapping(model.ImplicitlyCreatedDatasetCollectionInput, ), ) +simple_mapping(model.ImplicitCollectionJobs) + +# simple_mapping( +# model.ImplicitCollectionJobsHistoryDatasetCollectionAssociation, +# history_dataset_collection_associations=relation( +# model.HistoryDatasetCollectionAssociation, +# backref=backref("implicit_collection_jobs_association", uselist=False), +# uselist=True, +# ), +# ) + +simple_mapping( + model.ImplicitCollectionJobsJobAssociation, + implicit_collection_jobs=relation( + model.ImplicitCollectionJobs, + backref=backref("jobs", uselist=True), + uselist=False, + ), + job=relation( + model.Job, + backref=backref("implicit_collection_jobs_association", uselist=False), + uselist=False, + ), +) + mapper(model.JobParameter, model.JobParameter.table) mapper(model.JobExternalOutputMetadata, model.JobExternalOutputMetadata.table, properties=dict( @@ -2202,6 +2244,16 @@ simple_mapping(model.HistoryDatasetCollectionAssociation, model.ImplicitlyCreatedDatasetCollectionInput.table.c.dataset_collection_id)), backref="dataset_collection", ), + implicit_collection_jobs=relation( + model.ImplicitCollectionJobs, + backref=backref("history_dataset_collection_associations", uselist=True), + uselist=False, + ), + job=relation( + model.Job, + backref=backref("history_dataset_collection_associations", uselist=True), + uselist=False, + ), tags=relation(model.HistoryDatasetCollectionTagAssociation, order_by=model.HistoryDatasetCollectionTagAssociation.table.c.id, backref='dataset_collections'), diff --git a/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py b/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py index 1b83a49aa31..ead24344502 100644 --- a/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py +++ b/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py @@ -1,5 +1,5 @@ """ -Migration script for workflow request tables. +Migration script for collections and workflows connections. """ from __future__ import print_function @@ -79,6 +79,26 @@ def get_new_tables(): Column("job_id", Integer, ForeignKey("job.id"), index=True, nullable=False), ) + implicit_collection_jobs_table = Table( + "implicit_collection_jobs", metadata, + Column("id", Integer, primary_key=True), + Column("populated_state", TrimmedString(64), default='new', nullable=False), + ) + + # implicit_collection_jobs_history_dataset_collection_association_table = Table( + # "implicit_collection_jobs_dataset_collection_association", metadata, + # Column("id", Integer, primary_key=True), + # Column("history_dataset_collection_association_id", Integer, ForeignKey("history_dataset_collection_association_id.id"), index=True, nullable=False), + # ) + + implicit_collection_jobs_job_association_table = Table( + "implicit_collection_jobs_job_association", metadata, + Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), index=True), + Column("id", Integer, primary_key=True), + Column("job_id", Integer, ForeignKey("job.id"), index=True), # Consider making this nullable... + Column("order_index", Integer, nullable=False), + ) + tables = OrderedDict() tables["workflow_invocation_step"] = workflow_invocation_step_table tables["workflow_invocation_output_dataset_association"] = workflow_invocation_output_dataset_association_table @@ -86,6 +106,9 @@ def get_new_tables(): tables["workflow_invocation_step_output_dataset_association"] = workflow_invocation_step_output_dataset_association_table tables["workflow_invocation_step_output_dataset_collection_association"] = workflow_invocation_step_output_dataset_collection_association_table tables["workflow_invocation_step_job_association"] = workflow_invocation_step_job_association_table + tables["implicit_collection_jobs"] = implicit_collection_jobs_table + # tables["implicit_collection_jobs_history_dataset_collection_association"] = implicit_collection_jobs_history_dataset_collection_association_table + tables["implicit_collection_jobs_job_association"] = implicit_collection_jobs_job_association_table return tables @@ -157,6 +180,24 @@ def upgrade(migrate_engine): "WHERE job_id is not NULL " migrate_engine.execute(cmd) + if migrate_engine.name in ['postgres', 'postgresql']: + implicit_collection_jobs_id_column = Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), nullable=True) + job_id_column = Column("job_id", Integer, ForeignKey("job.id"), nullable=True) + else: + implicit_collection_jobs_id_column = Column("implicit_collection_jobs_id", Integer, nullable=True) + job_id_column = Column("job_id", Integer, nullable=True) + __add_column(implicit_collection_jobs_id_column, "history_dataset_collection_association", metadata) + __add_column(job_id_column, "history_dataset_collection_association", metadata) + # TODO: matching drop... steal from 0131 + + +def __add_column(column, table_name, metadata, **kwds): + try: + table = Table(table_name, metadata, autoload=True) + column.create(table, **kwds) + except Exception: + log.exception("Adding column %s failed.", column) + def downgrade(migrate_engine): metadata.bind = migrate_engine diff --git a/lib/galaxy/tools/actions/__init__.py b/lib/galaxy/tools/actions/__init__.py index 14bd8f54452..bfc467fb727 100644 --- a/lib/galaxy/tools/actions/__init__.py +++ b/lib/galaxy/tools/actions/__init__.py @@ -598,6 +598,7 @@ class DefaultToolAction(object): job.add_implicit_output_dataset_collection(name, dataset_collection) for name, dataset_collection_instance in out_collection_instances.items(): job.add_output_dataset_collection(name, dataset_collection_instance) + dataset_collection_instance.job = job def _check_input_data_access(self, trans, job, inp_data, current_user_roles): access_timer = ExecutionTimer() diff --git a/lib/galaxy/tools/execute.py b/lib/galaxy/tools/execute.py index 7e96ce8c914..ce04b02670c 100644 --- a/lib/galaxy/tools/execute.py +++ b/lib/galaxy/tools/execute.py @@ -42,9 +42,9 @@ def execute(trans, tool, mapping_params, history, rerun_remap_job_id=None, colle all_jobs_timer = ExecutionTimer() if invocation_step is None: - execution_tracker = ToolExecutionTracker(tool, mapping_params, collection_info) + execution_tracker = ToolExecutionTracker(trans, tool, mapping_params, collection_info) else: - execution_tracker = WorkflowStepExecutionTracker(tool, mapping_params, collection_info, invocation_step, job_callback=job_callback) + execution_tracker = WorkflowStepExecutionTracker(trans, tool, mapping_params, collection_info, invocation_step, job_callback=job_callback) app = trans.app execution_cache = ToolExecutionCache(trans) @@ -77,7 +77,7 @@ def execute(trans, tool, mapping_params, history, rerun_remap_job_id=None, colle history ) - execution_tracker.ensure_implicit_collections_populated(trans, history, mapping_params.param_template) + execution_tracker.ensure_implicit_collections_populated(history, mapping_params.param_template) config = app.config burst_at = getattr(config, 'tool_submission_burst_at', 10) burst_threads = getattr(config, 'tool_submission_burst_threads', 1) @@ -139,8 +139,9 @@ class ExecutionSlice(object): class ExecutionTracker(object): - def __init__(self, tool, mapping_params, collection_info): + def __init__(self, trans, tool, mapping_params, collection_info): # Known ahead of time... + self.trans = trans self.tool = tool self.mapping_params = mapping_params self.collection_info = collection_info @@ -247,23 +248,26 @@ class ExecutionTracker(object): mapped_output_structure = mapping_structure.multiply(output_structure, uninitialized=True) return mapped_output_structure - def ensure_implicit_collections_populated(self, trans, history, params): + def ensure_implicit_collections_populated(self, history, params): if not self.collection_info: return - history = history or self.tool.get_default_history_by_trans(trans) + history = history or self.tool.get_default_history_by_trans(self.trans) # params = param_combinations[0] if param_combinations else mapping_params.param_template - self.precreate_output_collections(trans, history, params) + self.precreate_output_collections(history, params) - def precreate_output_collections(self, trans, history, params): + def precreate_output_collections(self, history, params): # params is just one sample tool param execution with parallelized # collection replaced with a specific dataset. Need to replace this # with the collection and wrap everything up so can evaluate output # label. + trans = self.trans params.update(self.collection_info.collections) # Replace datasets with source collections for labelling outputs. collection_instances = {} implicit_inputs = self.implicit_inputs + + implicit_collection_jobs = model.ImplicitCollectionJobs() for output_name, output in self.tool.outputs.items(): if filter_output(output, self.example_params): continue @@ -277,6 +281,7 @@ class ExecutionTracker(object): implicit_output_name=output_name, structure=effective_structure, ) + collection_instance.implicit_collection_jobs = implicit_collection_jobs collection_instances[output_name] = collection_instance trans.sa_session.add(collection_instance) # Needed to flush the association created just above with @@ -291,11 +296,19 @@ class ExecutionTracker(object): # if you are mapping a list over a tool that dynamically generates lists - we won't actually # know the structure of the inner list until after its job is complete. if self.failed_jobs > 0: - for implicit_collection in self.implicit_collections.values(): + for i, implicit_collection in enumerate(self.implicit_collections.values()): + if i == 0: + implicit_collection_jobs = implicit_collection.implicit_collection_jobs + implicit_collection_jobs.populated_state = "failed" + trans.sa_session.add(implicit_collection_jobs) implicit_collection.collection.handle_population_failed("One or more jobs failed during dataset initialization.") trans.sa_session.add(implicit_collection.collection) else: - for implicit_collection in self.implicit_collections.values(): + for i, implicit_collection in enumerate(self.implicit_collections.values()): + if i == 0: + implicit_collection_jobs = implicit_collection.implicit_collection_jobs + implicit_collection_jobs.populated_state = "ok" + trans.sa_session.add(implicit_collection_jobs) implicit_collection.collection.finalize() trans.sa_session.add(implicit_collection.collection) trans.sa_session.flush() @@ -326,16 +339,25 @@ class ExecutionTracker(object): for job_output in job.output_dataset_collection_instances: self.output_collections.append((job_output.name, job_output.dataset_collection_instance)) if self.implicit_collections: + implicit_collection_jobs = None for output_name, collection_instance in self.implicit_collections.items(): job.add_output_dataset_collection(output_name, collection_instance) + if implicit_collection_jobs is None: + implicit_collection_jobs = collection_instance.implicit_collection_jobs + + job_assoc = model.ImplicitCollectionJobsJobAssociation() + job_assoc.order_index = execution_slice.job_index + job_assoc.implicit_collection_jobs = implicit_collection_jobs + job_assoc.job_id = job.id + self.trans.sa_session.add(job_assoc) # Seperate these because workflows need to track their jobs belong to the invocation # in the database immediately and they can be recovered. class ToolExecutionTracker(ExecutionTracker): - def __init__(self, tool, mapping_params, collection_info): - super(ToolExecutionTracker, self).__init__(tool, mapping_params, collection_info) + def __init__(self, trans, tool, mapping_params, collection_info): + super(ToolExecutionTracker, self).__init__(trans, tool, mapping_params, collection_info) # New to track these things for tool output API response in the tool case, # in the workflow case we just write stuff to the database and forget about @@ -363,17 +385,17 @@ class ToolExecutionTracker(ExecutionTracker): class WorkflowStepExecutionTracker(ExecutionTracker): - def __init__(self, tool, mapping_params, collection_info, invocation_step, job_callback): - super(WorkflowStepExecutionTracker, self).__init__(tool, mapping_params, collection_info) + def __init__(self, trans, tool, mapping_params, collection_info, invocation_step, job_callback): + super(WorkflowStepExecutionTracker, self).__init__(trans, tool, mapping_params, collection_info) self.invocation_step = invocation_step self.job_callback = job_callback def record_success(self, execution_slice, job, outputs): super(WorkflowStepExecutionTracker, self).record_success(execution_slice, job, outputs) - job_assoc = model.WorkflowInvocationStepJobAssociation() - job_assoc.index = execution_slice.job_index - job_assoc.workflow_invocation_step = self.invocation_step - job_assoc.job_id = job.id + # job_assoc = model.WorkflowInvocationStepJobAssociation() + # job_assoc.index = execution_slice.job_index + # job_assoc.workflow_invocation_step = self.invocation_step + # job_assoc.job_id = job.id self.job_callback(job) def new_collection_execution_slices(self): @@ -390,13 +412,13 @@ class WorkflowStepExecutionTracker(ExecutionTracker): continue yield ExecutionSlice(job_index, param_combination, dataset_collection_elements) - def ensure_implicit_collections_populated(self, trans, history, params): - if not self. collection_info: + def ensure_implicit_collections_populated(self, history, params): + if not self.collection_info: return - history = history or self.tool.get_default_history_by_trans(trans) + history = history or self.tool.get_default_history_by_trans(self.trans) if self.invocation_step.is_new: - self.precreate_output_collections(trans, history, params) + self.precreate_output_collections(history, params) else: collections = {} for output_assoc in self.invocation_step.output_dataset_collections: diff --git a/lib/galaxy/webapps/galaxy/api/history_contents.py b/lib/galaxy/webapps/galaxy/api/history_contents.py index 87954ef4542..f6fc932b5a0 100644 --- a/lib/galaxy/webapps/galaxy/api/history_contents.py +++ b/lib/galaxy/webapps/galaxy/api/history_contents.py @@ -21,6 +21,7 @@ from galaxy.managers.collections_util import ( dictify_dataset_collection_instance, get_hda_and_element_identifiers ) +from galaxy.managers.jobs import summarize_jobs_to_dict from galaxy.util.json import safe_dumps from galaxy.util.streamball import StreamBall from galaxy.web import ( @@ -154,6 +155,39 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary elif contents_type == 'dataset_collection': return self.__show_dataset_collection(trans, id, history_id, **kwd) + @expose_api_anonymous + def show_jobs_summary(self, trans, id, history_id, **kwd): + """ + * GET /api/histories/{history_id}/contents/{type}/{id}/jobs_summary + return detailed information about an HDA or HDCAs jobs + .. note:: Anonymous users are allowed to get their current history contents + + :type id: str + :param id: the encoded id of the HDA to return + :type history_id: str + :param history_id: encoded id string of the HDA's or the HDCA's History + + :rtype: dict + :returns: dictionary containing + """ + contents_type = self.__get_contents_type(trans, kwd) + # At most one of job or implicit_collection_jobs should be found. + job = None + implicit_collection_jobs = None + if contents_type == 'dataset': + hda = self.hda_manager.get_accessible(self.decode_id(id), trans.user) + job = hda.creating_job + elif contents_type == 'dataset_collection': + dataset_collection_instance = self.__get_accessible_collection(trans, id, history_id) + job_source_type = dataset_collection_instance.job_source_type + if job_source_type == "Job": + job = dataset_collection_instance.job + elif job_source_type == "ImplicitCollectionJobs": + implicit_collection_jobs = dataset_collection_instance.implicit_collection_jobs + + assert job is None or implicit_collection_jobs is None + return self.encode_all_ids(trans, summarize_jobs_to_dict(trans.sa_session, job or implicit_collection_jobs)) + def __get_contents_type(self, trans, kwd): contents_type = kwd.get('type', 'dataset') if contents_type not in ['dataset', 'dataset_collection']: diff --git a/lib/galaxy/webapps/galaxy/buildapp.py b/lib/galaxy/webapps/galaxy/buildapp.py index 5b8aadb97c0..cab77e6cfb4 100644 --- a/lib/galaxy/webapps/galaxy/buildapp.py +++ b/lib/galaxy/webapps/galaxy/buildapp.py @@ -363,6 +363,10 @@ def populate_api_routes(webapp, app): action='download_dataset_collection', conditions=dict(method=["GET"])) + webapp.mapper.connect("/api/histories/{history_id}/contents/{type:%s}s/{id}/jobs_summary" % "|".join(valid_history_contents_types), + action="show_jobs_summary", + controller='history_contents', + conditions=dict(method=["GET"])) # ---- visualizations registry ---- generic template renderer # @deprecated: this route should be considered deprecated webapp.add_route('/visualization/show/{visualization_name}', controller='visualization', action='render', visualization_name=None) diff --git a/scripts/db_shell.py b/scripts/db_shell.py index f7e10d61086..c896ec04af6 100644 --- a/scripts/db_shell.py +++ b/scripts/db_shell.py @@ -23,6 +23,7 @@ from six import string_types from sqlalchemy import * # noqa from sqlalchemy.orm import * # noqa from sqlalchemy.exc import * # noqa +from sqlalchemy.sql import label # noqa sys.path.insert(1, os.path.abspath(os.path.join(os.path.dirname(__file__), os.pardir, 'lib'))) diff --git a/test/api/test_history_contents.py b/test/api/test_history_contents.py index 5cb2822556b..1d9d77da2f2 100644 --- a/test/api/test_history_contents.py +++ b/test/api/test_history_contents.py @@ -6,9 +6,11 @@ from requests import delete, put from base import api # noqa: I100 from base.populators import ( # noqa: I100 + DatasetPopulator, DatasetCollectionPopulator, LibraryPopulator, - TestsDatasets + skip_without_tool, + TestsDatasets, ) @@ -18,6 +20,7 @@ class HistoryContentsApiTestCase(api.ApiTestCase, TestsDatasets): def setUp(self): super(HistoryContentsApiTestCase, self).setUp() self.history_id = self._new_history() + self.dataset_populator = DatasetPopulator(self.galaxy_interactor) self.dataset_collection_populator = DatasetCollectionPopulator(self.galaxy_interactor) self.library_populator = LibraryPopulator(self) @@ -177,6 +180,38 @@ class HistoryContentsApiTestCase(api.ApiTestCase, TestsDatasets): dataset_collection = show_response.json() assert dataset_collection["deleted"] + @skip_without_tool("collection_creates_list") + def test_jobs_summary_simple_hdca(self): + create_response = self.dataset_collection_populator.create_list_in_history(self.history_id, contents=["a\nb\nc\nd", "e\nf\ng\nh"]) + hdca_id = create_response.json()["id"] + run = self.dataset_populator.run_collection_creates_list(self.history_id, hdca_id) + collections = run['output_collections'] + collection = collections[0] + jobs_summary_url = "histories/%s/contents/dataset_collections/%s/jobs_summary" % (self.history_id, collection["id"]) + jobs_summary_response = self._get(jobs_summary_url) + self._assert_status_code_is(jobs_summary_response, 200) + jobs_summary = jobs_summary_response.json() + self._assert_has_keys(jobs_summary, "populated_state", "states") + + @skip_without_tool("cat1") + def test_jobs_summary_implicit_hdca(self): + create_response = self.dataset_collection_populator.create_pair_in_history(self.history_id, contents=["123", "456"]) + hdca_id = create_response.json()["id"] + inputs = { + "input1": {'batch': True, 'values': [{'src': 'hdca', 'id': hdca_id}]}, + } + run = self.dataset_populator.run_tool("cat1", inputs=inputs, history_id=self.history_id) + self.dataset_populator.wait_for_history_jobs(self.history_id) + collections = run['implicit_collections'] + collection = collections[0] + jobs_summary_url = "histories/%s/contents/dataset_collections/%s/jobs_summary" % (self.history_id, collection["id"]) + jobs_summary_response = self._get(jobs_summary_url) + self._assert_status_code_is(jobs_summary_response, 200) + jobs_summary = jobs_summary_response.json() + self._assert_has_keys(jobs_summary, "populated_state", "states") + states = jobs_summary["states"] + assert states.get("ok") == 2, states + def test_dataset_collection_hide_originals(self): payload = self.dataset_collection_populator.create_pair_payload( self.history_id, diff --git a/test/api/test_tools.py b/test/api/test_tools.py index 49a819909d6..264cd0d64f7 100644 --- a/test/api/test_tools.py +++ b/test/api/test_tools.py @@ -433,20 +433,15 @@ class ToolsTestCase(api.ApiTestCase): @skip_without_tool("collection_creates_list") def test_list_collection_output(self): - history_id = self.dataset_populator.new_history() - create_response = self.dataset_collection_populator.create_list_in_history(history_id, contents=["a\nb\nc\nd", "e\nf\ng\nh"]) - hdca_id = create_response.json()["id"] - inputs = { - "input1": {"src": "hdca", "id": hdca_id}, - } - # TODO: real problem here - shouldn't have to have this wait. - self.dataset_populator.wait_for_history(history_id, assert_ok=True) - create = self._run("collection_creates_list", history_id, inputs, assert_ok=True) - output_collection = self._assert_one_job_one_collection_run(create) - element0, element1 = self._assert_elements_are(output_collection, "data1", "data2") - self.dataset_populator.wait_for_history(history_id, assert_ok=True) - self._verify_element(history_id, element0, contents="identifier is data1\n", file_ext="txt") - self._verify_element(history_id, element1, contents="identifier is data2\n", file_ext="txt") + with self.dataset_populator.test_history() as history_id: + create_response = self.dataset_collection_populator.create_list_in_history(history_id, contents=["a\nb\nc\nd", "e\nf\ng\nh"]) + hdca_id = create_response.json()["id"] + create = self.dataset_populator.run_collection_creates_list(history_id, hdca_id) + output_collection = self._assert_one_job_one_collection_run(create) + element0, element1 = self._assert_elements_are(output_collection, "data1", "data2") + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + self._verify_element(history_id, element0, contents="identifier is data1\n", file_ext="txt") + self._verify_element(history_id, element1, contents="identifier is data2\n", file_ext="txt") @skip_without_tool("collection_creates_list_2") def test_list_collection_output_format_source(self): diff --git a/test/base/populators.py b/test/base/populators.py index ae98d60d15e..16bccfcbd9e 100644 --- a/test/base/populators.py +++ b/test/base/populators.py @@ -291,6 +291,13 @@ class BaseDatasetPopulator(object): assert details_response.status_code == 200, details_response.content return details_response.json() + def run_collection_creates_list(self, history_id, hdca_id): + inputs = { + "input1": {"src": "hdca", "id": hdca_id}, + } + self.wait_for_history(history_id, assert_ok=True) + return self.run_tool("collection_creates_list", inputs, history_id) + def __history_content_id(self, history_id, wait=True, **kwds): if wait: assert_ok = kwds.get("assert_ok", True) From e82012bde154c456f24743f3b61c0ed9ffc7eca2 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Wed, 25 Oct 2017 09:05:32 -0400 Subject: [PATCH 15/22] Undo splitting WorkflowInvocationStep... ... this concept of ImplicitCollectionJobs makes more sense for tracking collections of implicitly created jobs. --- lib/galaxy/jobs/actions/post.py | 12 +- lib/galaxy/model/__init__.py | 22 +- lib/galaxy/model/mapping.py | 24 +- .../versions/0136_record_workflow_outputs.py | 236 ++++++++---------- lib/galaxy/tools/execute.py | 18 +- 5 files changed, 143 insertions(+), 169 deletions(-) diff --git a/lib/galaxy/jobs/actions/post.py b/lib/galaxy/jobs/actions/post.py index 138d4b8005b..267a765e66f 100644 --- a/lib/galaxy/jobs/actions/post.py +++ b/lib/galaxy/jobs/actions/post.py @@ -269,10 +269,10 @@ class DeleteIntermediatesAction(DefaultJobAction): # concurrently, sometimes non-terminal steps won't be cleaned up # because of the lag in job state updates. sa_session.flush() - if not job.workflow_invocation_step_assoc.workflow_invocation_step: + if not job.workflow_invocation_step: log.debug("This job is not part of a workflow invocation, delete intermediates aborted.") return - wfi = job.workflow_invocation_step_assoc.workflow_invocation_step.workflow_invocation + wfi = job.workflow_invocation_step.workflow_invocation sa_session.refresh(wfi) if wfi.active: log.debug("Workflow still scheduling so new jobs may appear, skipping deletion of intermediate files.") @@ -285,9 +285,9 @@ class DeleteIntermediatesAction(DefaultJobAction): jobs_to_check = [] for wfi_step in wfi_steps: sa_session.refresh(wfi_step) - wfi_step_job_assocs = wfi_step.jobs - if wfi_step_job_assocs: - jobs_to_check.extend(map(lambda j: j.job, wfi_step_job_assocs)) + wfi_step_job = wfi_step.job + if wfi_step_job: + jobs_to_check.append(wfi_step_job) else: log.debug("No job found yet for wfi_step %s, (step %s)" % (wfi_step, wfi_step.workflow_step)) for j2c in jobs_to_check: @@ -302,7 +302,7 @@ class DeleteIntermediatesAction(DefaultJobAction): for (input_dataset, creating_job) in creating_jobs: sa_session.refresh(creating_job) sa_session.refresh(input_dataset) - for input_dataset in [x.dataset for (x, creating_job) in creating_jobs if creating_job.workflow_invocation_step_assoc and creating_job.workflow_invocation_step_assoc.workflow_invocation_step.workflow_invocation == wfi]: + for input_dataset in [x.dataset for (x, creating_job) in creating_jobs if creating_job.workflow_invocation_step and creating_job.workflow_invocation_step.workflow_invocation == wfi]: # note that the above input_dataset is a reference to a # job.input_dataset.dataset at this point safe_to_delete = True diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index a495bb65a75..98679c442be 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -803,9 +803,8 @@ class Job(object, JobLike, UsesCreateAndUpdateTime, Dictifiable): def set_final_state(self, final_state): self.set_state(final_state) - workflow_invocation_step_assoc = self.workflow_invocation_step_assoc - if workflow_invocation_step_assoc: - workflow_invocation_step_assoc.workflow_invocation_step.update() + if self.workflow_invocation_step: + self.workflow_invocation_step.update() def get_destination_configuration(self, config, key, default=None): """ Get a destination parameter that can be defaulted back @@ -1070,6 +1069,10 @@ class ImplicitCollectionJobs(object): self.id = id self.populated_state = populated_state or ImplicitCollectionJobs.populated_states.NEW + @property + def job_list(self): + return [icjja.job for icjja in self.jobs] + class ImplicitCollectionJobsJobAssociation(object): @@ -4159,12 +4162,12 @@ class WorkflowInvocation(object, UsesCreateAndUpdateTime, Dictifiable): inputs = {} for step in self.steps: if step.workflow_step.type == 'tool': - for step_job_assoc in step.jobs: + for job in step.jobs: for step_input in step.workflow_step.input_connections: output_step_type = step_input.output_step.type if output_step_type in ['data_input', 'data_collection_input']: src = "hda" if output_step_type == 'data_input' else 'hdca' - for job_input in step_job_assoc.job.input_datasets: + for job_input in job.input_datasets: if job_input.name == step_input.input_name: inputs[str(step_input.output_step.order_index)] = { "id": job_input.dataset_id, "src": src, @@ -4268,6 +4271,15 @@ class WorkflowInvocationStep(object, Dictifiable): else: raise Exception("Uknown output type encountered") + @property + def jobs(self): + if self.job: + return [self.job] + elif self.implicit_collection_jobs: + return self.implicit_collection_jobs.job_list + else: + return [] + def to_dict(self, view='collection', value_mapper=None): rval = super(WorkflowInvocationStep, self).to_dict(view=view, value_mapper=value_mapper) rval['order_index'] = self.workflow_step.order_index diff --git a/lib/galaxy/model/mapping.py b/lib/galaxy/model/mapping.py index f53799db844..895da39a79b 100644 --- a/lib/galaxy/model/mapping.py +++ b/lib/galaxy/model/mapping.py @@ -919,18 +919,10 @@ model.WorkflowInvocationStep.table = Table( Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True, nullable=False), Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True, nullable=False), Column("state", TrimmedString(64), index=True), + Column("job_id", Integer, ForeignKey("job.id"), index=True, nullable=True), + Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), index=True, nullable=True), Column("action", JSONType, nullable=True)) - -model.WorkflowInvocationStepJobAssociation.table = Table( - "workflow_invocation_step_job_association", metadata, - Column("id", Integer, primary_key=True), - Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True, nullable=False), - Column("order_index", Integer, nullable=True), # recovering partially complete WorkflowInvocationSteps requires knowing which jobs have been scheduled - Column("job_id", Integer, ForeignKey("job.id"), index=True, nullable=False), -) - - model.WorkflowInvocationOutputDatasetAssociation.table = Table( "workflow_invocation_output_dataset_association", metadata, Column("id", Integer, primary_key=True), @@ -2420,16 +2412,10 @@ mapper(model.WorkflowInvocationToSubworkflowInvocationAssociation, model.Workflo workflow_step=relation(model.WorkflowStep), )) - -simple_mapping(model.WorkflowInvocationStepJobAssociation, - workflow_invocation_step=relation(model.WorkflowInvocationStep, backref="jobs"), - job=relation(model.Job, - backref=backref('workflow_invocation_step_assoc', - uselist=False))) - - simple_mapping(model.WorkflowInvocationStep, - workflow_step=relation(model.WorkflowStep)) + workflow_step=relation(model.WorkflowStep), + job=relation(model.Job, backref=backref('workflow_invocation_step', uselist=False), uselist=False), + implicit_collection_jobs=relation(model.ImplicitCollectionJobs, backref=backref('workflow_invocation_step', uselist=False), uselist=False),) simple_mapping(model.WorkflowRequestInputParameter, diff --git a/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py b/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py index ead24344502..1c23ed1c56c 100644 --- a/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py +++ b/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py @@ -8,10 +8,9 @@ import logging from collections import OrderedDict -from migrate.changeset.constraint import ForeignKeyConstraint -from sqlalchemy import Column, DateTime, ForeignKey, Integer, MetaData, String, Table +from sqlalchemy import Column, ForeignKey, Integer, MetaData, String, Table -from galaxy.model.custom_types import JSONType, TrimmedString +from galaxy.model.custom_types import TrimmedString now = datetime.datetime.utcnow @@ -20,92 +19,93 @@ log = logging.getLogger(__name__) metadata = MetaData() +workflow_invocation_output_dataset_association_table = Table( + "workflow_invocation_output_dataset_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True), + Column("workflow_step_id", Integer, ForeignKey("workflow_step.id")), + Column("dataset_id", Integer, ForeignKey("history_dataset_association.id"), index=True), + Column("workflow_output_id", Integer, ForeignKey("workflow_output.id")), +) + +workflow_invocation_output_dataset_collection_association_table = Table( + "workflow_invocation_output_dataset_collection_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True), + Column("workflow_step_id", Integer, ForeignKey("workflow_step.id")), + Column("dataset_collection_id", Integer, ForeignKey("history_dataset_collection_association.id"), index=True), + Column("workflow_output_id", Integer, ForeignKey("workflow_output.id")), +) + +workflow_invocation_step_output_dataset_association_table = Table( + "workflow_invocation_step_output_dataset_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True), + Column("dataset_id", Integer, ForeignKey("history_dataset_association.id"), index=True), + Column("output_name", String(255), nullable=True), +) + +workflow_invocation_step_output_dataset_collection_association_table = Table( + "workflow_invocation_step_output_dataset_collection_association", metadata, + Column("id", Integer, primary_key=True), + Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True), + Column("workflow_step_id", Integer, ForeignKey("workflow_step.id")), + Column("dataset_collection_id", Integer, ForeignKey("history_dataset_collection_association.id"), index=True), + Column("output_name", String(255), nullable=True), +) + +# workflow_invocation_step_table = Table( +# "workflow_invocation_step", metadata, +# Column("id", Integer, primary_key=True), +# Column("create_time", DateTime, default=now), +# Column("update_time", DateTime, default=now, onupdate=now), +# Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True, nullable=False), +# Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True, nullable=False), +# Column("action", JSONType, nullable=True), +# Column("state", TrimmedString(64), default="new"), +# ) + +# workflow_invocation_step_job_association_table = Table( +# "workflow_invocation_step_job_association", metadata, +# Column("id", Integer, primary_key=True), +# Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True, nullable=False), +# Column("order_index", Integer, nullable=True), +# Column("job_id", Integer, ForeignKey("job.id"), index=True, nullable=False), +# ) + +implicit_collection_jobs_table = Table( + "implicit_collection_jobs", metadata, + Column("id", Integer, primary_key=True), + Column("populated_state", TrimmedString(64), default='new', nullable=False), +) + +# implicit_collection_jobs_history_dataset_collection_association_table = Table( +# "implicit_collection_jobs_dataset_collection_association", metadata, +# Column("id", Integer, primary_key=True), +# Column("history_dataset_collection_association_id", Integer, ForeignKey("history_dataset_collection_association_id.id"), index=True, nullable=False), +# ) + +implicit_collection_jobs_job_association_table = Table( + "implicit_collection_jobs_job_association", metadata, + Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), index=True), + Column("id", Integer, primary_key=True), + Column("job_id", Integer, ForeignKey("job.id"), index=True), # Consider making this nullable... + Column("order_index", Integer, nullable=False), +) + + def get_new_tables(): # Normally we define this globally in the file, but we need to delay the # reading of existing tables because an existing workflow_invocation_step # table exists that we want to recreate. - workflow_invocation_output_dataset_association_table = Table( - "workflow_invocation_output_dataset_association", metadata, - Column("id", Integer, primary_key=True), - Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True), - Column("workflow_step_id", Integer, ForeignKey("workflow_step.id")), - Column("dataset_id", Integer, ForeignKey("history_dataset_association.id"), index=True), - Column("workflow_output_id", Integer, ForeignKey("workflow_output.id")), - ) - - workflow_invocation_output_dataset_collection_association_table = Table( - "workflow_invocation_output_dataset_collection_association", metadata, - Column("id", Integer, primary_key=True), - Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True), - Column("workflow_step_id", Integer, ForeignKey("workflow_step.id")), - Column("dataset_collection_id", Integer, ForeignKey("history_dataset_collection_association.id"), index=True), - Column("workflow_output_id", Integer, ForeignKey("workflow_output.id")), - ) - - workflow_invocation_step_output_dataset_association_table = Table( - "workflow_invocation_step_output_dataset_association", metadata, - Column("id", Integer, primary_key=True), - Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True), - Column("dataset_id", Integer, ForeignKey("history_dataset_association.id"), index=True), - Column("output_name", String(255), nullable=True), - ) - - workflow_invocation_step_output_dataset_collection_association_table = Table( - "workflow_invocation_step_output_dataset_collection_association", metadata, - Column("id", Integer, primary_key=True), - Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True), - Column("workflow_step_id", Integer, ForeignKey("workflow_step.id")), - Column("dataset_collection_id", Integer, ForeignKey("history_dataset_collection_association.id"), index=True), - Column("output_name", String(255), nullable=True), - ) - - workflow_invocation_step_table = Table( - "workflow_invocation_step", metadata, - Column("id", Integer, primary_key=True), - Column("create_time", DateTime, default=now), - Column("update_time", DateTime, default=now, onupdate=now), - Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True, nullable=False), - Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True, nullable=False), - Column("action", JSONType, nullable=True), - Column("state", TrimmedString(64), default="new"), - ) - - workflow_invocation_step_job_association_table = Table( - "workflow_invocation_step_job_association", metadata, - Column("id", Integer, primary_key=True), - Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True, nullable=False), - Column("order_index", Integer, nullable=True), - Column("job_id", Integer, ForeignKey("job.id"), index=True, nullable=False), - ) - - implicit_collection_jobs_table = Table( - "implicit_collection_jobs", metadata, - Column("id", Integer, primary_key=True), - Column("populated_state", TrimmedString(64), default='new', nullable=False), - ) - - # implicit_collection_jobs_history_dataset_collection_association_table = Table( - # "implicit_collection_jobs_dataset_collection_association", metadata, - # Column("id", Integer, primary_key=True), - # Column("history_dataset_collection_association_id", Integer, ForeignKey("history_dataset_collection_association_id.id"), index=True, nullable=False), - # ) - - implicit_collection_jobs_job_association_table = Table( - "implicit_collection_jobs_job_association", metadata, - Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), index=True), - Column("id", Integer, primary_key=True), - Column("job_id", Integer, ForeignKey("job.id"), index=True), # Consider making this nullable... - Column("order_index", Integer, nullable=False), - ) - tables = OrderedDict() - tables["workflow_invocation_step"] = workflow_invocation_step_table + # tables["workflow_invocation_step"] = workflow_invocation_step_table tables["workflow_invocation_output_dataset_association"] = workflow_invocation_output_dataset_association_table tables["workflow_invocation_output_dataset_collection_association"] = workflow_invocation_output_dataset_collection_association_table tables["workflow_invocation_step_output_dataset_association"] = workflow_invocation_step_output_dataset_association_table tables["workflow_invocation_step_output_dataset_collection_association"] = workflow_invocation_step_output_dataset_collection_association_table - tables["workflow_invocation_step_job_association"] = workflow_invocation_step_job_association_table + # tables["workflow_invocation_step_job_association"] = workflow_invocation_step_job_association_table tables["implicit_collection_jobs"] = implicit_collection_jobs_table # tables["implicit_collection_jobs_history_dataset_collection_association"] = implicit_collection_jobs_history_dataset_collection_association_table tables["implicit_collection_jobs_job_association"] = implicit_collection_jobs_job_association_table @@ -117,21 +117,6 @@ def upgrade(migrate_engine): metadata.bind = migrate_engine print(__doc__) - LegacyWorkflowInvocationStep_table = Table("workflow_invocation_step", metadata, autoload=True) - ExistingWorkflowInvocation_table = Table("workflow_invocation", metadata, autoload=True) - - cons = ForeignKeyConstraint([LegacyWorkflowInvocationStep_table.c.workflow_invocation_id], [ExistingWorkflowInvocation_table.c.id]) - cons.drop() - - for index in LegacyWorkflowInvocationStep_table.indexes: - index.drop() - - LegacyWorkflowInvocationStep_table.rename("workflow_invocation_step_premigrate135") - # Try to deregister that workflow_invocation_step to work around some caching problems - # it seems. - LegacyWorkflowInvocationStep_table.deregister() - metadata._remove_table("workflow_invocation_step", metadata.schema) - metadata.reflect() tables = get_new_tables() for table in tables.values(): @@ -145,41 +130,8 @@ def upgrade(migrate_engine): else: raise Exception("Unhandled database type") - # Skips action - since I can't aggregate (sql needs a RANDOM(col)) that and it is only used by optional - # beta extensions. - cmd = \ - "INSERT INTO workflow_invocation_step " + \ - "(id, create_time, update_time, workflow_invocation_id, workflow_step_id, action, state)" + \ - "SELECT " + \ - nextval('workflow_invocation_step') + " AS id, " \ - "MIN(create_time) AS create_time, " + \ - "MAX(update_time) AS update_time, " + \ - "workflow_invocation_step_premigrate135.workflow_invocation_id AS workflow_invocation_id, " +\ - "workflow_invocation_step_premigrate135.workflow_step_id AS workflow_step_id, " + \ - "NULL AS action, " + \ - "'scheduled' AS state " + \ - "FROM workflow_invocation_step_premigrate135 " + \ - "WHERE workflow_invocation_step_premigrate135.workflow_step_id IS NOT NULL AND workflow_invocation_step_premigrate135.workflow_invocation_id IS NOT NULL " +\ - "GROUP BY workflow_invocation_step_premigrate135.workflow_invocation_id, workflow_invocation_step_premigrate135.workflow_step_id " + \ - "" - migrate_engine.execute(cmd) - - cmd = \ - "INSERT INTO workflow_invocation_step_job_association " + \ - "(id, workflow_invocation_step_id, order_index, job_id) " + \ - "SELECT " + \ - nextval('workflow_invocation_step_job_association') + " AS id, " \ - "workflow_invocation_step.id AS workflow_invocation_step_id, " + \ - "NULL AS order_index, " + \ - "job_id AS job_id " + \ - "FROM workflow_invocation_step_premigrate135 " + \ - "LEFT JOIN workflow_invocation_step on (" + \ - " workflow_invocation_step.workflow_invocation_id = workflow_invocation_step_premigrate135.workflow_invocation_id " + \ - " AND workflow_invocation_step.workflow_step_id = workflow_invocation_step_premigrate135.workflow_step_id" + \ - ") " + \ - "WHERE job_id is not NULL " - migrate_engine.execute(cmd) - + # Set default for creation to scheduled, actual mapping has new as default. + workflow_invocation_step_state_column = Column("state", TrimmedString(64), default="scheduled") if migrate_engine.name in ['postgres', 'postgresql']: implicit_collection_jobs_id_column = Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), nullable=True) job_id_column = Column("job_id", Integer, ForeignKey("job.id"), nullable=True) @@ -188,6 +140,11 @@ def upgrade(migrate_engine): job_id_column = Column("job_id", Integer, nullable=True) __add_column(implicit_collection_jobs_id_column, "history_dataset_collection_association", metadata) __add_column(job_id_column, "history_dataset_collection_association", metadata) + + implicit_collection_jobs_id_column = Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), nullable=True) + __add_column(implicit_collection_jobs_id_column, "workflow_invocation_step", metadata) + __add_column(workflow_invocation_step_state_column, "workflow_invocation_step", metadata) + # TODO: matching drop... steal from 0131 @@ -199,17 +156,26 @@ def __add_column(column, table_name, metadata, **kwds): log.exception("Adding column %s failed.", column) +def __drop_column(column_name, table_name, metadata): + try: + table = Table(table_name, metadata, autoload=True) + getattr(table.c, column_name).drop() + except Exception: + log.exception("Dropping column %s failed.", column_name) + + def downgrade(migrate_engine): metadata.bind = migrate_engine metadata.reflect() - tables = get_new_tables() - for table in tables.values(): - __drop(table) + __drop_column("implicit_collection_jobs_id", "history_dataset_collection_association", metadata) + __drop_column("job_id", "history_dataset_collection_association", metadata) + __drop_column("implicit_collection_jobs_id", "workflow_invocation_step", metadata) + __drop_column("state", "workflow_invocation_step", metadata) - # Drop new workflow invocation step and job association table and restore legacy data. - LegacyWorkflowInvocationStep_table = Table("workflow_invocation_step_premigrate135", metadata, autoload=True) - LegacyWorkflowInvocationStep_table.rename("workflow_invocation_step") + tables = get_new_tables() + for table in reversed(tables.values()): + __drop(table) def __create(table): diff --git a/lib/galaxy/tools/execute.py b/lib/galaxy/tools/execute.py index ce04b02670c..80130a51b1d 100644 --- a/lib/galaxy/tools/execute.py +++ b/lib/galaxy/tools/execute.py @@ -8,6 +8,8 @@ import itertools import logging from threading import Thread +import six + from six.moves.queue import Queue from galaxy import model @@ -289,6 +291,14 @@ class ExecutionTracker(object): trans.sa_session.flush() self.implicit_collections = collection_instances + @property + def implicit_collection_jobs(self): + # TODO: refactor to track this properly maybe? + if self.implicit_collections: + return six.next(six.itervalues(self.implicit_collections)).implicit_collection_jobs + else: + return None + def finalize_dataset_collections(self, trans): # TODO: this probably needs to be reworked some, we should have the collection methods # return a list of changed objects to add to the session and flush and we should only @@ -392,10 +402,10 @@ class WorkflowStepExecutionTracker(ExecutionTracker): def record_success(self, execution_slice, job, outputs): super(WorkflowStepExecutionTracker, self).record_success(execution_slice, job, outputs) - # job_assoc = model.WorkflowInvocationStepJobAssociation() - # job_assoc.index = execution_slice.job_index - # job_assoc.workflow_invocation_step = self.invocation_step - # job_assoc.job_id = job.id + if self.collection_info: + self.invocation_step.implicit_collection_jobs = self.implicit_collection_jobs + else: + self.invocation_step.job = job self.job_callback(job) def new_collection_execution_slices(self): From 0dceb5a553d589324d4a9a97ec5f7629bc379b48 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Wed, 8 Nov 2017 08:44:56 -0500 Subject: [PATCH 16/22] Add element count to DatasetCollection objects. There are no places where this count is not set atomically in one thread so this should not have any performance downsides versus the older approach of calculating it on the fly (which was very expensive). --- lib/galaxy/dataset_collections/builder.py | 1 + lib/galaxy/managers/collections.py | 1 + lib/galaxy/managers/collections_util.py | 1 + lib/galaxy/managers/hdcas.py | 12 ++---- lib/galaxy/model/__init__.py | 1 + lib/galaxy/model/mapping.py | 1 + .../versions/0136_record_workflow_outputs.py | 38 +++++-------------- test/api/test_dataset_collections.py | 4 +- 8 files changed, 19 insertions(+), 40 deletions(-) diff --git a/lib/galaxy/dataset_collections/builder.py b/lib/galaxy/dataset_collections/builder.py index b32b26ae312..bdb21e75cbf 100644 --- a/lib/galaxy/dataset_collections/builder.py +++ b/lib/galaxy/dataset_collections/builder.py @@ -24,6 +24,7 @@ def set_collection_elements(dataset_collection, type, dataset_instances): element_index += 1 dataset_collection.elements = elements + dataset_collection.element_count = element_index return dataset_collection diff --git a/lib/galaxy/managers/collections.py b/lib/galaxy/managers/collections.py index 356729593a9..e823b0f44ca 100644 --- a/lib/galaxy/managers/collections.py +++ b/lib/galaxy/managers/collections.py @@ -76,6 +76,7 @@ class DatasetCollectionManager(object): ) elements.append(element) dataset_collection.elements = elements + dataset_collection.element_count = len(elements) return dataset_collection diff --git a/lib/galaxy/managers/collections_util.py b/lib/galaxy/managers/collections_util.py index 190cbd2b929..65101597d77 100644 --- a/lib/galaxy/managers/collections_util.py +++ b/lib/galaxy/managers/collections_util.py @@ -130,6 +130,7 @@ def dictify_element(element): child_collection = element.child_collection object_detials["elements"] = [dictify_element(_) for _ in child_collection.elements] object_detials["populated"] = child_collection.populated + object_detials["element_count"] = child_collection.element_count else: object_detials = None diff --git a/lib/galaxy/managers/hdcas.py b/lib/galaxy/managers/hdcas.py index 877c27390ab..ea3cdb1706c 100644 --- a/lib/galaxy/managers/hdcas.py +++ b/lib/galaxy/managers/hdcas.py @@ -119,6 +119,7 @@ class DCSerializer(base.ModelSerializer): 'collection_type', 'populated_state', 'populated_state_message', + 'element_count', ]) self.add_view('detailed', [ 'populated', @@ -130,7 +131,6 @@ class DCSerializer(base.ModelSerializer): self.serializers.update({ 'model_class' : lambda *a, **c: 'DatasetCollection', 'elements' : self.serialize_elements, - 'element_count' : self.serialize_element_count, }) def serialize_elements(self, item, key, **context): @@ -140,14 +140,6 @@ class DCSerializer(base.ModelSerializer): returned.append(serialized) return returned - def serialize_element_count(self, item, key, **context): - """Return the count of elements for this collection.""" - # TODO: app.model.context -> session - # TODO: to the container interface (dataset_collection_contents) - return (self.app.model.context.query(model.DatasetCollectionElement) - .filter(model.DatasetCollectionElement.dataset_collection_id == item.id) - .count()) - class DCASerializer(base.ModelSerializer): """ @@ -165,6 +157,7 @@ class DCASerializer(base.ModelSerializer): 'collection_type', 'populated_state', 'populated_state_message', + 'element_count', ]) self.add_view('detailed', [ 'populated', @@ -223,6 +216,7 @@ class HDCASerializer( 'collection_type', 'populated_state', 'populated_state_message', + 'element_count', 'job_source_id', 'job_source_type', diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 98679c442be..e968423a6a7 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -3353,6 +3353,7 @@ class DatasetCollectionInstance(object, HasName): populated=self.populated, populated_state=self.collection.populated_state, populated_state_message=self.collection.populated_state_message, + element_count=self.collection.element_count, type="collection", # contents type (distinguished from file or folder (in case of library)) ) diff --git a/lib/galaxy/model/mapping.py b/lib/galaxy/model/mapping.py index 895da39a79b..ad4aec98d3b 100644 --- a/lib/galaxy/model/mapping.py +++ b/lib/galaxy/model/mapping.py @@ -716,6 +716,7 @@ model.DatasetCollection.table = Table( Column("collection_type", Unicode(255), nullable=False), Column("populated_state", TrimmedString(64), default='ok', nullable=False), Column("populated_state_message", TEXT), + Column("element_count", Integer, nullable=True), Column("create_time", DateTime, default=now), Column("update_time", DateTime, default=now, onupdate=now)) diff --git a/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py b/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py index 1c23ed1c56c..4f739a2e7c3 100644 --- a/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py +++ b/lib/galaxy/model/migrate/versions/0136_record_workflow_outputs.py @@ -54,37 +54,12 @@ workflow_invocation_step_output_dataset_collection_association_table = Table( Column("output_name", String(255), nullable=True), ) -# workflow_invocation_step_table = Table( -# "workflow_invocation_step", metadata, -# Column("id", Integer, primary_key=True), -# Column("create_time", DateTime, default=now), -# Column("update_time", DateTime, default=now, onupdate=now), -# Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True, nullable=False), -# Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True, nullable=False), -# Column("action", JSONType, nullable=True), -# Column("state", TrimmedString(64), default="new"), -# ) - -# workflow_invocation_step_job_association_table = Table( -# "workflow_invocation_step_job_association", metadata, -# Column("id", Integer, primary_key=True), -# Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True, nullable=False), -# Column("order_index", Integer, nullable=True), -# Column("job_id", Integer, ForeignKey("job.id"), index=True, nullable=False), -# ) - implicit_collection_jobs_table = Table( "implicit_collection_jobs", metadata, Column("id", Integer, primary_key=True), Column("populated_state", TrimmedString(64), default='new', nullable=False), ) -# implicit_collection_jobs_history_dataset_collection_association_table = Table( -# "implicit_collection_jobs_dataset_collection_association", metadata, -# Column("id", Integer, primary_key=True), -# Column("history_dataset_collection_association_id", Integer, ForeignKey("history_dataset_collection_association_id.id"), index=True, nullable=False), -# ) - implicit_collection_jobs_job_association_table = Table( "implicit_collection_jobs_job_association", metadata, Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), index=True), @@ -100,14 +75,11 @@ def get_new_tables(): # table exists that we want to recreate. tables = OrderedDict() - # tables["workflow_invocation_step"] = workflow_invocation_step_table tables["workflow_invocation_output_dataset_association"] = workflow_invocation_output_dataset_association_table tables["workflow_invocation_output_dataset_collection_association"] = workflow_invocation_output_dataset_collection_association_table tables["workflow_invocation_step_output_dataset_association"] = workflow_invocation_step_output_dataset_association_table tables["workflow_invocation_step_output_dataset_collection_association"] = workflow_invocation_step_output_dataset_collection_association_table - # tables["workflow_invocation_step_job_association"] = workflow_invocation_step_job_association_table tables["implicit_collection_jobs"] = implicit_collection_jobs_table - # tables["implicit_collection_jobs_history_dataset_collection_association"] = implicit_collection_jobs_history_dataset_collection_association_table tables["implicit_collection_jobs_job_association"] = implicit_collection_jobs_job_association_table return tables @@ -138,14 +110,21 @@ def upgrade(migrate_engine): else: implicit_collection_jobs_id_column = Column("implicit_collection_jobs_id", Integer, nullable=True) job_id_column = Column("job_id", Integer, nullable=True) + dataset_collection_element_count_column = Column("element_count", Integer, nullable=True) + __add_column(implicit_collection_jobs_id_column, "history_dataset_collection_association", metadata) __add_column(job_id_column, "history_dataset_collection_association", metadata) + __add_column(dataset_collection_element_count_column, "dataset_collection", metadata) implicit_collection_jobs_id_column = Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), nullable=True) __add_column(implicit_collection_jobs_id_column, "workflow_invocation_step", metadata) __add_column(workflow_invocation_step_state_column, "workflow_invocation_step", metadata) - # TODO: matching drop... steal from 0131 + cmd = \ + "UPDATE dataset_collection SET element_count = " + \ + "(SELECT (CASE WHEN count(*) > 0 THEN count(*) ELSE 0 END) FROM dataset_collection_element WHERE " + \ + "dataset_collection_element.dataset_collection_id = dataset_collection.id)" + migrate_engine.execute(cmd) def __add_column(column, table_name, metadata, **kwds): @@ -172,6 +151,7 @@ def downgrade(migrate_engine): __drop_column("job_id", "history_dataset_collection_association", metadata) __drop_column("implicit_collection_jobs_id", "workflow_invocation_step", metadata) __drop_column("state", "workflow_invocation_step", metadata) + __drop_column("element_count", "dataset_collection", metadata) tables = get_new_tables() for table in reversed(tables.values()): diff --git a/test/api/test_dataset_collections.py b/test/api/test_dataset_collections.py index f8db6028570..69bdbbdc8ff 100644 --- a/test/api/test_dataset_collections.py +++ b/test/api/test_dataset_collections.py @@ -87,7 +87,7 @@ class DatasetCollectionApiTestCase(api.ApiTestCase): pair_1_element = returned_collections[0] self._assert_has_keys(pair_1_element, "element_index") pair_1_object = pair_1_element["object"] - self._assert_has_keys(pair_1_object, "collection_type", "elements") + self._assert_has_keys(pair_1_object, "collection_type", "elements", "element_count") self.assertEquals(pair_1_object["collection_type"], "paired") self.assertEquals(pair_1_object["populated"], True) pair_elements = pair_1_object["elements"] @@ -195,7 +195,7 @@ class DatasetCollectionApiTestCase(api.ApiTestCase): def _check_create_response(self, create_response): self._assert_status_code_is(create_response, 200) dataset_collection = create_response.json() - self._assert_has_keys(dataset_collection, "elements", "url", "name", "collection_type") + self._assert_has_keys(dataset_collection, "elements", "url", "name", "collection_type", "element_count") return dataset_collection def _download_dataset_collection(self, history_id, hdca_id): From ca09e803aeaf856556433e69bf5a48f8bab60c5d Mon Sep 17 00:00:00 2001 From: John Chilton Date: Wed, 8 Nov 2017 14:11:55 -0500 Subject: [PATCH 17/22] Dataset collection state UX test case. --- test/api/test_tools.py | 6 +- test/base/populators.py | 8 ++ .../collection_creates_dynamic_nested.xml | 8 +- ...collection_creates_dynamic_nested_fail.xml | 2 +- test/galaxy_selenium/navigates_galaxy.py | 3 + test/galaxy_selenium/navigation.yml | 21 ++++ test/selenium_tests/framework.py | 17 +++- .../test_history_copy_elements.py | 40 ++++++++ .../selenium_tests/test_history_multi_view.py | 24 +++++ .../test_history_panel_collections.py | 99 +++++++++++++++++++ 10 files changed, 218 insertions(+), 10 deletions(-) create mode 100644 test/selenium_tests/test_history_copy_elements.py create mode 100644 test/selenium_tests/test_history_multi_view.py create mode 100644 test/selenium_tests/test_history_panel_collections.py diff --git a/test/api/test_tools.py b/test/api/test_tools.py index 264cd0d64f7..a1ea61d38fb 100644 --- a/test/api/test_tools.py +++ b/test/api/test_tools.py @@ -206,11 +206,7 @@ class ToolsTestCase(api.ApiTestCase): with self.dataset_populator.test_history() as history_id: history_id = self.dataset_populator.new_history() ok_hdca_id = self.dataset_collection_populator.create_list_in_history(history_id, contents=["0", "1", "0", "1"]).json()["id"] - exit_code_inputs = { - "input": {'batch': True, 'values': [{"src": "hdca", "id": ok_hdca_id}]}, - } - response = self._run("exit_code_from_file", history_id, exit_code_inputs, assert_ok=False).json() - self.dataset_populator.wait_for_history(history_id, assert_ok=False) + response = self.dataset_populator.run_exit_code_from_file(history_id, ok_hdca_id) mixed_implicit_collections = response["implicit_collections"] self.assertEquals(len(mixed_implicit_collections), 1) diff --git a/test/base/populators.py b/test/base/populators.py index 16bccfcbd9e..9ff96bb4bb4 100644 --- a/test/base/populators.py +++ b/test/base/populators.py @@ -298,6 +298,14 @@ class BaseDatasetPopulator(object): self.wait_for_history(history_id, assert_ok=True) return self.run_tool("collection_creates_list", inputs, history_id) + def run_exit_code_from_file(self, history_id, hdca_id): + exit_code_inputs = { + "input": {'batch': True, 'values': [{"src": "hdca", "id": hdca_id}]}, + } + response = self.run_tool("exit_code_from_file", exit_code_inputs, history_id, assert_ok=False).json() + self.wait_for_history(history_id, assert_ok=False) + return response + def __history_content_id(self, history_id, wait=True, **kwds): if wait: assert_ok = kwds.get("assert_ok", True) diff --git a/test/functional/tools/collection_creates_dynamic_nested.xml b/test/functional/tools/collection_creates_dynamic_nested.xml index 68edc0cc34b..40a3f50247c 100644 --- a/test/functional/tools/collection_creates_dynamic_nested.xml +++ b/test/functional/tools/collection_creates_dynamic_nested.xml @@ -1,14 +1,16 @@ - + oe1_ie1.fq ; echo "B" > oe1_ie2.fq ; echo "C" > oe2_ie1.fq ; echo "D" > oe2_ie2.fq ; echo "E" > oe3_ie1.fq ; - echo "F" > oe3_ie2.fq - + echo "F" > oe3_ie2.fq ; + sleep '$sleep_time'; + ]]> + diff --git a/test/functional/tools/collection_creates_dynamic_nested_fail.xml b/test/functional/tools/collection_creates_dynamic_nested_fail.xml index 370e85f19b6..68a44bef693 100644 --- a/test/functional/tools/collection_creates_dynamic_nested_fail.xml +++ b/test/functional/tools/collection_creates_dynamic_nested_fail.xml @@ -12,7 +12,7 @@ - + \s*$)|\s+/g,"").split(",")):n=e,i="auto"===o.get("width")?n.length*o.get("defaultPixelsPerValue"):o.get("width"),"auto"===o.get("height")?o.get("composite")&&t.data(this,"_jqs_vcanvas")||(l=s.createElement("span"),l.innerHTML="a",a.html(l),r=t(l).innerHeight()||t(l).height(),t(l).remove(),l=null):r=o.get("height"),o.get("disableInteraction")?u=!1:(u=t.data(this,"_jqs_mhandler"),u?o.get("composite")||u.reset():(u=new m(this,o),t.data(this,"_jqs_mhandler",u))),o.get("composite")&&!t.data(this,"_jqs_vcanvas"))return void(t.data(this,"_jqs_errnotify")||(alert("Attempted to attach a composite sparkline to an element with no existing sparkline"),t.data(this,"_jqs_errnotify",!0)));c=new(t.fn.sparkline[o.get("type")])(this,n,o,i,r),c.render(),u&&u.registerSparkline(c)},t(this).html()&&!o.get("disableHiddenCheck")&&t(this).is(":hidden")||!t(this).parents("body").length){if(!o.get("composite")&&t.data(this,"_jqs_pending"))for(r=P.length;r;r--)P[r-1][0]==this&&P.splice(r-1,1);P.push([this,i]),t.data(this,"_jqs_pending",!0)}else i.call(this)})},t.fn.sparkline.defaults=e(),t.sparkline_display_visible=function(){var e,n,i,r=[];for(n=0,i=P.length;nthis.canvasWidth||n>this.canvasHeight||e<0||n<0?null:(i=this.getRegion(t,e,n),r!==i&&(void 0!==r&&o&&this.removeHighlight(),this.currentRegion=i,void 0!==i&&o&&this.renderHighlight(),!0))},clearRegionHighlight:function(){return void 0!==this.currentRegion&&(this.removeHighlight(),this.currentRegion=void 0,!0)},renderHighlight:function(){this.changeHighlight(!0)},removeHighlight:function(){this.changeHighlight(!1)},changeHighlight:function(t){},getCurrentRegionTooltip:function(){var e,n,r,o,s,a,l,u,c,h,d,p,f,g,v=this.options,m="",y=[];if(void 0===this.currentRegion)return"";if(e=this.getCurrentRegionFields(),d=v.get("tooltipFormatter"))return d(this,v,e);if(v.get("tooltipChartTitle")&&(m+='
'+v.get("tooltipChartTitle")+"
\n"),!(n=this.options.get("tooltipFormat")))return"";if(t.isArray(n)||(n=[n]),t.isArray(e)||(e=[e]),l=this.options.get("tooltipFormatFieldlist"),u=this.options.get("tooltipFormatFieldlistKey"),l&&u){for(c=[],a=e.length;a--;)h=e[a][u],-1!=(g=t.inArray(h,l))&&(c[g]=e[a]);e=c}for(r=n.length,f=e.length,a=0;a'+s+""));return y.length?m+y.join("\n"):""},getCurrentRegionFields:function(){},calcHighlightColor:function(t,e){var n,i,o,s,l=e.get("highlightColor"),u=e.get("highlightLighten");if(l)return l;if(u&&(n=/^#([0-9a-f])([0-9a-f])([0-9a-f])$/i.exec(t)||/^#([0-9a-f]{2})([0-9a-f]{2})([0-9a-f]{2})$/i.exec(t))){for(o=[],i=4===t.length?16:1,s=0;s<3;s++)o[s]=r(a.round(parseInt(n[s+1],16)*i*u),0,255);return"rgb("+o.join(",")+")"}return t}}),b={changeHighlight:function(e){var n,i=this.currentRegion,r=this.target,o=this.regionShapes[i];o&&(n=this.renderRegion(i,e),t.isArray(n)||t.isArray(o)?(r.replaceWithShapes(o,n),this.regionShapes[i]=t.map(n,function(t){return t.id})):(r.replaceWithShape(o,n),this.regionShapes[i]=n.id))},render:function(){var e,n,i,r,o=this.values,s=this.target,a=this.regionShapes;if(this.cls._super.render.call(this)){for(i=o.length;i--;)if(e=this.renderRegion(i))if(t.isArray(e)){for(n=[],r=e.length;r--;)e[r].append(),n.push(e[r].id);a[i]=n}else e.append(),a[i]=e.id;else a[i]=null;s.render()}}},t.fn.sparkline.line=w=n(t.fn.sparkline._base,{type:"line",init:function(t,e,n,i,r){w._super.init.call(this,t,e,n,i,r),this.vertices=[],this.regionMap=[],this.xvalues=[],this.yvalues=[],this.yminmax=[],this.hightlightSpotId=null,this.lastShapeId=null,this.initTarget()},getRegion:function(t,e,n){var i,r=this.regionMap;for(i=r.length;i--;)if(null!==r[i]&&e>=r[i][0]&&e<=r[i][1])return r[i][2]},getCurrentRegionFields:function(){var t=this.currentRegion;return{isNull:null===this.yvalues[t],x:this.xvalues[t],y:this.yvalues[t],color:this.options.get("lineColor"),fillColor:this.options.get("fillColor"),offset:t}},renderHighlight:function(){var t,e,n=this.currentRegion,i=this.target,r=this.vertices[n],o=this.options,s=o.get("spotRadius"),a=o.get("highlightSpotColor"),l=o.get("highlightLineColor");r&&(s&&a&&(t=i.drawCircle(r[0],r[1],s,void 0,a),this.highlightSpotId=t.id,i.insertAfterShape(this.lastShapeId,t)),l&&(e=i.drawLine(r[0],this.canvasTop,r[0],this.canvasTop+this.canvasHeight,l),this.highlightLineId=e.id,i.insertAfterShape(this.lastShapeId,e)))},removeHighlight:function(){var t=this.target;this.highlightSpotId&&(t.removeShapeId(this.highlightSpotId),this.highlightSpotId=null),this.highlightLineId&&(t.removeShapeId(this.highlightLineId),this.highlightLineId=null)},scanValues:function(){var t,e,n,i,r,o=this.values,s=o.length,l=this.xvalues,u=this.yvalues,c=this.yminmax;for(t=0;tthis.maxy&&(this.maxy=n)),void 0!==t.get("chartRangeMin")&&(t.get("chartRangeClip")||t.get("chartRangeMin")this.maxy)&&(this.maxy=t.get("chartRangeMax")),void 0!==t.get("chartRangeMinX")&&(t.get("chartRangeClipX")||t.get("chartRangeMinX")this.maxx)&&(this.maxx=t.get("chartRangeMaxX"))},drawNormalRange:function(t,e,n,i,r){var o=this.options.get("normalRangeMin"),s=this.options.get("normalRangeMax"),l=e+a.round(n-n*((s-this.miny)/r)),u=a.round(n*(s-o)/r);this.target.drawRect(t,l,i,u,void 0,this.options.get("normalRangeColor")).append()},render:function(){var e,n,i,r,o,s,l,u,c,h,d,p,f,g,m,y,b,_,x,C,k,S,T,A,E,O=this.options,j=this.target,D=this.canvasWidth,M=this.canvasHeight,P=this.vertices,N=O.get("spotRadius"),R=this.regionMap;if(w._super.render.call(this)&&(this.scanValues(),this.processRangeOptions(),T=this.xvalues,A=this.yvalues,this.yminmax.length&&!(this.yvalues.length<2))){for(r=o=0,e=this.maxx-this.minx==0?1:this.maxx-this.minx,n=this.maxy-this.miny==0?1:this.maxy-this.miny,i=this.yvalues.length-1,N&&(D<4*N||M<4*N)&&(N=0),N&&(k=O.get("highlightSpotColor")&&!O.get("disableInteraction"),(k||O.get("minSpotColor")||O.get("spotColor")&&A[i]===this.miny)&&(M-=a.ceil(N)),(k||O.get("maxSpotColor")||O.get("spotColor")&&A[i]===this.maxy)&&(M-=a.ceil(N),r+=a.ceil(N)),(k||(O.get("minSpotColor")||O.get("maxSpotColor"))&&(A[0]===this.miny||A[0]===this.maxy))&&(o+=a.ceil(N),D-=a.ceil(N)),(k||O.get("spotColor")||O.get("minSpotColor")||O.get("maxSpotColor")&&(A[i]===this.miny||A[i]===this.maxy))&&(D-=a.ceil(N))),M--,void 0===O.get("normalRangeMin")||O.get("drawNormalOnTop")||this.drawNormalRange(o,r,M,D,n),l=[],u=[l],g=m=null,y=A.length,E=0;Ethis.maxy&&(h=this.maxy),l.length||l.push([p,r+M]),s=[p,r+a.round(M-M*((h-this.miny)/n))],l.push(s),P.push(s));for(b=[],_=[],x=u.length,E=0;E2&&(l[0]=[l[0][0],l[1][1]]),b.push(l));for(x=_.length,E=0;E-1)||t.isArray(M))&&(A=!0,h&&(M=n[m]=u(M.split(":"))),M=c(M,null),d=a.min.apply(a,M),p=a.max.apply(a,M),dq&&(q=p));this.stacked=A,this.regionShapes={},this.barWidth=R,this.barSpacing=I,this.totalBarWidth=R+I,this.width=o=n.length*R+(n.length-1)*I,this.initTarget(),H&&(S=void 0===$?-1/0:$,T=void 0===L?1/0:L),g=[],f=A?[]:g;var W=[],B=[];for(m=0,y=n.length;m0&&(W[m]+=M),F<0&&q>0?M<0?B[m]+=a.abs(M):f[m]+=M:f[m]+=a.abs(M-(M<0?q:F)),g.push(M));else M=H?r(n[m],S,T):n[m],null!==(M=n[m]=l(M))&&g.push(M);this.max=k=a.max.apply(a,g),this.min=C=a.min.apply(a,g),this.stackMax=q=A?a.max.apply(a,W):k,this.stackMin=F=A?a.min.apply(a,g):C,void 0!==i.get("chartRangeMin")&&(i.get("chartRangeClip")||i.get("chartRangeMin")k)&&(k=i.get("chartRangeMax")),this.zeroAxis=w=i.get("zeroAxis",!0),x=C<=0&&k>=0&&w?0:0==w?C:C>0?C:k,this.xaxisOffset=x,b=A?a.max.apply(a,f)+a.max.apply(a,B):k-C,this.canvasHeightEf=w&&C<0?this.canvasHeight-2:this.canvasHeight-1,C=0?q:k,(P=(N-x)/b*this.canvasHeight)!==a.ceil(P)&&(this.canvasHeightEf-=2,P=a.ceil(P))):P=this.canvasHeight,this.yoffset=P,t.isArray(i.get("colorMap"))?(this.colorMapByIndex=i.get("colorMap"),this.colorMapByValue=null):(this.colorMapByIndex=null,this.colorMapByValue=i.get("colorMap"),this.colorMapByValue&&void 0===this.colorMapByValue.get&&(this.colorMapByValue=new v(this.colorMapByValue))),this.range=b},getRegion:function(t,e,n){var i=a.floor(e/this.totalBarWidth);return i<0||i>=this.values.length?void 0:i},getCurrentRegionFields:function(){var t,e,n=this.currentRegion,i=f(this.values[n]),r=[];for(e=i.length;e--;)t=i[e],r.push({isNull:null===t,value:t,color:this.calcColor(e,t,n),offset:n});return r},calcColor:function(e,n,i){var r,o,s=this.colorMapByIndex,a=this.colorMapByValue,l=this.options;return r=this.stacked?l.get("stackedBarColor"):n<0?l.get("negBarColor"):l.get("barColor"),0===n&&void 0!==l.get("zeroColor")&&(r=l.get("zeroColor")),a&&(o=a.get(n))?r=o:s&&s.length>i&&(r=s[i]),t.isArray(r)?r[e%r.length]:r},renderRegion:function(e,n){var i,r,o,s,l,u,c,h,p,f,g=this.values[e],v=this.options,m=this.xaxisOffset,y=[],b=this.range,w=this.stacked,_=this.target,x=e*this.totalBarWidth,C=this.canvasHeightEf,k=this.yoffset;if(g=t.isArray(g)?g:[g],c=g.length,h=g[0],s=d(null,g),f=d(m,g,!0),s)return v.get("nullColor")?(o=n?v.get("nullColor"):this.calcHighlightColor(v.get("nullColor"),v),i=k>0?k-1:k,_.drawRect(x,i,this.barWidth-1,0,o,o)):void 0;for(l=k,u=0;u0?a.floor(C*(a.abs(h-m)/b))+1:1,he?o[e]:i[e]<0?r.get("negBarColor"):i[e]>0?r.get("posBarColor"):r.get("zeroBarColor")},renderRegion:function(t,e){var n,i,r,o,s,l,u=this.values,c=this.options,h=this.target;if(n=h.pixelHeight,r=a.round(n/2),o=t*this.totalBarWidth,u[t]<0?(s=r,i=r-1):u[t]>0?(s=0,i=r-1):(s=r-1,i=2),null!==(l=this.calcColor(u[t],t)))return e&&(l=this.calcHighlightColor(l,c)),h.drawRect(o,s,this.barWidth-1,i-1,l,l)}}),t.fn.sparkline.discrete=C=n(t.fn.sparkline._base,b,{type:"discrete",init:function(e,n,i,r,o){C._super.init.call(this,e,n,i,r,o),this.regionShapes={},this.values=n=t.map(n,Number),this.min=a.min.apply(a,n),this.max=a.max.apply(a,n),this.range=this.max-this.min,this.width=r="auto"===i.get("width")?2*n.length:this.width,this.interval=a.floor(r/n.length),this.itemWidth=r/n.length,void 0!==i.get("chartRangeMin")&&(i.get("chartRangeClip")||i.get("chartRangeMin")this.max)&&(this.max=i.get("chartRangeMax")),this.initTarget(),this.target&&(this.lineHeight="auto"===i.get("lineHeight")?a.round(.3*this.canvasHeight):i.get("lineHeight"))},getRegion:function(t,e,n){return a.floor(e/this.itemWidth)},getCurrentRegionFields:function(){var t=this.currentRegion;return{isNull:void 0===this.values[t],value:this.values[t],offset:t}},renderRegion:function(t,e){var n,i,o,s,l=this.values,u=this.options,c=this.min,h=this.max,d=this.range,p=this.interval,f=this.target,g=this.canvasHeight,v=this.lineHeight,m=g-v;return i=r(l[t],c,h),s=t*p,n=a.round(m-m*((i-c)/d)),o=u.get("thresholdColor")&&i0)for(s=n.length;s--;)l+=n[s];this.total=l,this.initTarget(),this.radius=a.floor(a.min(this.canvasWidth,this.canvasHeight)/2)},getRegion:function(t,e,n){var i=this.target.getShapeAt(t,e,n);return void 0!==i&&void 0!==this.shapes[i]?this.shapes[i]:void 0},getCurrentRegionFields:function(){var t=this.currentRegion;return{isNull:void 0===this.values[t],value:this.values[t],percent:this.values[t]/this.total*100,color:this.options.get("sliceColors")[t%this.options.get("sliceColors").length],offset:t}},changeHighlight:function(t){var e=this.currentRegion,n=this.renderSlice(e,t),i=this.valueShapes[e];delete this.shapes[i],this.target.replaceWithShape(i,n),this.valueShapes[e]=n.id,this.shapes[n.id]=e},renderSlice:function(t,e){var n,i,r,o,s,l=this.target,u=this.options,c=this.radius,h=u.get("borderWidth"),d=u.get("offset"),p=2*a.PI,f=this.values,g=this.total,v=d?2*a.PI*(d/360):0;for(o=f.length,r=0;r0&&(i=v+p*(f[r]/g)),t===r)return s=u.get("sliceColors")[r%u.get("sliceColors").length],e&&(s=this.calcHighlightColor(s,u)),l.drawPieSlice(c,c,c-h,n,i,void 0,s);v=i}},render:function(){var t,e,n=this.target,i=this.values,r=this.options,o=this.radius,s=r.get("borderWidth");if(S._super.render.call(this)){for(s&&n.drawCircle(o,o,a.floor(o-s/2),r.get("borderColor"),void 0,s).append(),e=i.length;e--;)i[e]&&(t=this.renderSlice(e).append(),this.valueShapes[e]=t.id,this.shapes[t.id]=e);n.render()}}}),t.fn.sparkline.box=T=n(t.fn.sparkline._base,{type:"box",init:function(e,n,i,r,o){T._super.init.call(this,e,n,i,r,o),this.values=t.map(n,Number),this.width="auto"===i.get("width")?"4.0em":r,this.initTarget(),this.values.length||(this.disabled=1)},getRegion:function(){return 1},getCurrentRegionFields:function(){var t=[{field:"lq",value:this.quartiles[0]},{field:"med",value:this.quartiles[1]},{field:"uq",value:this.quartiles[2]}];return void 0!==this.loutlier&&t.push({field:"lo",value:this.loutlier}),void 0!==this.routlier&&t.push({field:"ro",value:this.routlier}),void 0!==this.lwhisker&&t.push({field:"lw",value:this.lwhisker}),void 0!==this.rwhisker&&t.push({field:"rw",value:this.rwhisker}),t},render:function(){var t,e,n,i,r,s,l,u,c,h,d,p=this.target,f=this.values,g=f.length,v=this.options,m=this.canvasWidth,y=this.canvasHeight,b=void 0===v.get("chartRangeMin")?a.min.apply(a,f):v.get("chartRangeMin"),w=void 0===v.get("chartRangeMax")?a.max.apply(a,f):v.get("chartRangeMax"),_=0;if(T._super.render.call(this)){if(v.get("raw"))v.get("showOutliers")&&f.length>5?(e=f[0],t=f[1],i=f[2],r=f[3],s=f[4],l=f[5],u=f[6]):(t=f[0],i=f[1],r=f[2],s=f[3],l=f[4]);else if(f.sort(function(t,e){return t-e}),i=o(f,1),r=o(f,2),s=o(f,3),n=s-i,v.get("showOutliers")){for(t=l=void 0,c=0;ci-n*v.get("outlierIQR")&&(t=f[c]),f[c]l&&p.drawCircle((u-b)*d+_,y/2,v.get("spotRadius"),v.get("outlierLineColor"),v.get("outlierFillColor")).append()),p.drawRect(a.round((i-b)*d+_),a.round(.1*y),a.round((s-i)*d),a.round(.8*y),v.get("boxLineColor"),v.get("boxFillColor")).append(),p.drawLine(a.round((t-b)*d+_),a.round(y/2),a.round((i-b)*d+_),a.round(y/2),v.get("lineColor")).append(),p.drawLine(a.round((t-b)*d+_),a.round(y/4),a.round((t-b)*d+_),a.round(y-y/4),v.get("whiskerColor")).append(),p.drawLine(a.round((l-b)*d+_),a.round(y/2),a.round((s-b)*d+_),a.round(y/2),v.get("lineColor")).append(),p.drawLine(a.round((l-b)*d+_),a.round(y/4),a.round((l-b)*d+_),a.round(y-y/4),v.get("whiskerColor")).append(),p.drawLine(a.round((r-b)*d+_),a.round(.1*y),a.round((r-b)*d+_),a.round(.9*y),v.get("medianColor")).append(),v.get("target")&&(h=a.ceil(v.get("spotRadius")),p.drawLine(a.round((v.get("target")-b)*d+_),a.round(y/2-h),a.round((v.get("target")-b)*d+_),a.round(y/2+h),v.get("targetColor")).append(),p.drawLine(a.round((v.get("target")-b)*d+_-h),a.round(y/2),a.round((v.get("target")-b)*d+_+h),a.round(y/2),v.get("targetColor")).append()),p.render()}}}),O=n({init:function(t,e,n,i){this.target=t,this.id=e,this.type=n,this.args=i},append:function(){return this.target.appendShape(this),this}}),j=n({_pxregex:/(\d+)(px)?\s*$/i,init:function(e,n,i){e&&(this.width=e,this.height=n,this.target=i,this.lastShapeId=null,i[0]&&(i=i[0]),t.data(i,"_jqs_vcanvas",this))},drawLine:function(t,e,n,i,r,o){return this.drawShape([[t,e],[n,i]],r,o)},drawShape:function(t,e,n,i){return this._genShape("Shape",[t,e,n,i])},drawCircle:function(t,e,n,i,r,o){return this._genShape("Circle",[t,e,n,i,r,o])},drawPieSlice:function(t,e,n,i,r,o,s){return this._genShape("PieSlice",[t,e,n,i,r,o,s])},drawRect:function(t,e,n,i,r,o){return this._genShape("Rect",[t,e,n,i,r,o])},getElement:function(){return this.canvas},getLastShapeId:function(){return this.lastShapeId},reset:function(){alert("reset not implemented")},_insert:function(e,n){t(n).html(e)},_calculatePixelDims:function(e,n,i){var r;r=this._pxregex.exec(n),this.pixelHeight=r?r[1]:t(i).height(),r=this._pxregex.exec(e),this.pixelWidth=r?r[1]:t(i).width()},_genShape:function(t,e){var n=R++;return e.unshift(n),new O(this,n,t,e)},appendShape:function(t){alert("appendShape not implemented")},replaceWithShape:function(t,e){alert("replaceWithShape not implemented")},insertAfterShape:function(t,e){alert("insertAfterShape not implemented")},removeShapeId:function(t){alert("removeShapeId not implemented")},getShapeAt:function(t,e,n){alert("getShapeAt not implemented")},render:function(){alert("render not implemented")}}),D=n(j,{init:function(e,n,i,r){D._super.init.call(this,e,n,i),this.canvas=s.createElement("canvas"),i[0]&&(i=i[0]),t.data(i,"_jqs_vcanvas",this),t(this.canvas).css({display:"inline-block",width:e,height:n,verticalAlign:"top"}),this._insert(this.canvas,i),this._calculatePixelDims(e,n,this.canvas),this.canvas.width=this.pixelWidth,this.canvas.height=this.pixelHeight,this.interact=r,this.shapes={},this.shapeseq=[],this.currentTargetShapeId=void 0,t(this.canvas).css({width:this.pixelWidth,height:this.pixelHeight})},_getContext:function(t,e,n){var i=this.canvas.getContext("2d");return void 0!==t&&(i.strokeStyle=t),i.lineWidth=void 0===n?1:n,void 0!==e&&(i.fillStyle=e),i},reset:function(){this._getContext().clearRect(0,0,this.pixelWidth,this.pixelHeight),this.shapes={},this.shapeseq=[],this.currentTargetShapeId=void 0},_drawShape:function(t,e,n,i,r){var o,s,a=this._getContext(n,i,r);for(a.beginPath(),a.moveTo(e[0][0]+.5,e[0][1]+.5),o=1,s=e.length;o',this.canvas.insertAdjacentHTML("beforeEnd",r),this.group=t(this.canvas).children()[0],this.rendered=!1,this.prerender=""},_drawShape:function(t,e,n,i,r){var o,s,a,l,u,c,h=[];for(c=0,u=e.length;c '},_drawCircle:function(t,e,n,i,r,o,s){var a,l;return e-=i,n-=i,a=void 0===r?' stroked="false" ':' strokeWeight="'+s+'px" strokeColor="'+r+'" ',l=void 0===o?' filled="false"':' fillColor="'+o+'" filled="true" ',''},_drawPieSlice:function(t,e,n,i,r,o,s,l){var u,c,h,d,p,f,g;if(r===o)return"";if(o-r==2*a.PI&&(r=0,o=2*a.PI),c=e+a.round(a.cos(r)*i),h=n+a.round(a.sin(r)*i),d=e+a.round(a.cos(o)*i),p=n+a.round(a.sin(o)*i),c===d&&h===p){if(o-r ')},_drawRect:function(t,e,n,i,r,o,s){return this._drawShape(t,[[e,n],[e,n+r],[e+i,n+r],[e+i,n],[e,n]],o,s)},reset:function(){this.group.innerHTML=""},appendShape:function(t){var e=this["_draw"+t.type].apply(this,t.args);return this.rendered?this.group.insertAdjacentHTML("beforeEnd",e):this.prerender+=e,this.lastShapeId=t.id,t.id},replaceWithShape:function(e,n){var i=t("#jqsshape"+e),r=this["_draw"+n.type].apply(this,n.args);i[0].outerHTML=r},replaceWithShapes:function(e,n){var i,r=t("#jqsshape"+e[0]),o="",s=n.length;for(i=0;i=0;t--)e=b.__jstorage_meta.PubSub[t],e[0]>E&&(n=e[0],d(e[1],e[2]));E=n}}function d(t,e){if(A[t])for(var n=0,i=A[t].length;n=4;)n=255&t.charCodeAt(o)|(255&t.charCodeAt(++o))<<8|(255&t.charCodeAt(++o))<<16|(255&t.charCodeAt(++o))<<24,n=1540483477*(65535&n)+((1540483477*(n>>>16)&65535)<<16),n^=n>>>24,n=1540483477*(65535&n)+((1540483477*(n>>>16)&65535)<<16),r=1540483477*(65535&r)+((1540483477*(r>>>16)&65535)<<16)^n,i-=4,++o;switch(i){case 3:r^=(255&t.charCodeAt(o+2))<<16;case 2:r^=(255&t.charCodeAt(o+1))<<8;case 1:r^=255&t.charCodeAt(o),r=1540483477*(65535&r)+((1540483477*(r>>>16)&65535)<<16)}return r^=r>>>13,r=1540483477*(65535&r)+((1540483477*(r>>>16)&65535)<<16),(r^=r>>>15)>>>0}var v=t||window.$||(window.$={}),m={parse:window.JSON&&(window.JSON.parse||window.JSON.decode)||String.prototype.evalJSON&&function(t){return String(t).evalJSON()}||v.parseJSON||v.evalJSON,stringify:Object.toJSON||window.JSON&&(window.JSON.stringify||window.JSON.encode)||v.toJSON};if(!("parse"in m&&"stringify"in m))throw new Error("No JSON support found, include //cdnjs.cloudflare.com/ajax/libs/json2/20110223/json2.js to page");var y,b={__jstorage_meta:{CRC32:{}}},w={jStorage:"{}"},_=null,x=0,C=!1,k={},S=!1,T=0,A={},E=+new Date,O={isXML:function(t){var e=(t?t.ownerDocument||t:0).documentElement;return!!e&&"HTML"!==e.nodeName},encode:function(t){if(!this.isXML(t))return!1;try{return(new XMLSerializer).serializeToString(t)}catch(e){try{return t.xml}catch(t){}}return!1},decode:function(t){var e,n="DOMParser"in window&&(new DOMParser).parseFromString||window.ActiveXObject&&function(t){var e=new ActiveXObject("Microsoft.XMLDOM");return e.async="false",e.loadXML(t),e};return!!n&&(e=n.call("DOMParser"in window&&new DOMParser||window,t,"text/xml"),!!this.isXML(e)&&e)}};v.jStorage={version:"0.4.4",set:function(t,e,n){if(u(t),n=n||{},void 0===e)return this.deleteKey(t),e;if(O.isXML(e))e={_is_xml:!0,xml:O.encode(e)};else{if("function"==typeof e)return;e&&"object"==typeof e&&(e=m.parse(m.stringify(e)))}return b[t]=e,b.__jstorage_meta.CRC32[t]="2."+g(m.stringify(e),2538058380),this.setTTL(t,n.TTL||0),o(t,"updated"),e},get:function(t,e){return u(t),t in b?b[t]&&"object"==typeof b[t]&&b[t]._is_xml?O.decode(b[t].xml):b[t]:void 0===e?null:e},deleteKey:function(t){return u(t),t in b&&(delete b[t],"object"==typeof b.__jstorage_meta.TTL&&t in b.__jstorage_meta.TTL&&delete b.__jstorage_meta.TTL[t],delete b.__jstorage_meta.CRC32[t],l(),s(),o(t,"deleted"),!0)},setTTL:function(t,e){var n=+new Date;return u(t),e=Number(e)||0,t in b&&(b.__jstorage_meta.TTL||(b.__jstorage_meta.TTL={}),e>0?b.__jstorage_meta.TTL[t]=n+e:delete b.__jstorage_meta.TTL[t],l(),c(),s(),!0)},getTTL:function(t){var e=+new Date;return u(t),t in b&&b.__jstorage_meta.TTL&&b.__jstorage_meta.TTL[t]?b.__jstorage_meta.TTL[t]-e||0:0},flush:function(){return b={__jstorage_meta:{CRC32:{}}},l(),s(),o(null,"flushed"),!0},storageObj:function(){function t(){}return t.prototype=b,new t},index:function(){var t,e=[];for(t in b)b.hasOwnProperty(t)&&"__jstorage_meta"!=t&&e.push(t);return e},storageSize:function(){return x},currentBackend:function(){return C},storageAvailable:function(){return!!C},listenKeyChange:function(t,e){u(t),k[t]||(k[t]=[]),k[t].push(e)},stopListening:function(t,e){if(u(t),k[t]){if(!e)return void delete k[t];for(var n=k[t].length-1;n>=0;n--)k[t][n]==e&&k[t].splice(n,1)}},subscribe:function(t,e){if(!(t=(t||"").toString()))throw new TypeError("Channel not defined");A[t]||(A[t]=[]),A[t].push(e)},publish:function(t,e){if(!(t=(t||"").toString()))throw new TypeError("Channel not defined");f(t,e)},reInit:function(){e()}},function(){var t=!1;if("localStorage"in window)try{window.localStorage.setItem("_tmptest","tmpval"),t=!0,window.localStorage.removeItem("_tmptest")}catch(t){}if(t)try{window.localStorage&&(w=window.localStorage,C="localStorage",T=w.jStorage_update)}catch(t){}else if("globalStorage"in window)try{window.globalStorage&&(w="localhost"==window.location.hostname?window.globalStorage["localhost.localdomain"]:window.globalStorage[window.location.hostname],C="globalStorage",T=w.jStorage_update)}catch(t){}else{if(_=document.createElement("link"),!_.addBehavior)return void(_=null);_.style.behavior="url(#default#userData)",document.getElementsByTagName("head")[0].appendChild(_);try{_.load("jStorage")}catch(t){_.setAttribute("jStorage","{}"),_.save("jStorage"),_.load("jStorage")}var e="{}";try{e=_.getAttribute("jStorage")}catch(t){}try{T=_.getAttribute("jStorage_update")}catch(t){}w.jStorage=e,C="userDataBehavior"}a(),c(),n(),h(),"addEventListener"in window&&window.addEventListener("pageshow",function(t){t.persisted&&i()},!1)}()}()}).call(e,n(0))},231:function(t,e,n){(function(t){!function(t){t.fn.extend({complexify:function(e,n){function i(t,e){for(var n=t.length-1;n>=0;n--)if(e[0]<=t.charCodeAt(n)&&t.charCodeAt(n)<=e[1])return e[1]-e[0]+1;return 0}function r(n){if("strict"===e.banMode){for(var i=0;i-1}function o(){var o=t(this).val(),u=0,c=!1;if(r(o))u=1;else for(var h=l.length-1;h>=0;h--)u+=i(o,l[h]);u=Math.log(Math.pow(u,o.length))*(1/e.strengthScaleFactor),c=u>s&&o.length>=e.minimumChars,u=u/a*100,u=u>100?100:u,n.call(this,c,u)}var s=49,a=120,l=[[32,32],[48,57],[65,90],[97,122],[33,47],[58,64],[91,96],[123,126],[128,255],[256,383],[384,591],[592,687],[688,767],[768,879],[880,1023],[1024,1279],[1328,1423],[1424,1535],[1536,1791],[1792,1871],[1920,1983],[2304,2431],[2432,2559],[2560,2687],[2688,2815],[2816,2943],[2944,3071],[3072,3199],[3200,3327],[3328,3455],[3456,3583],[3584,3711],[3712,3839],[3840,4095],[4096,4255],[4256,4351],[4352,4607],[4608,4991],[5024,5119],[5120,5759],[5760,5791],[5792,5887],[6016,6143],[6144,6319],[7680,7935],[7936,8191],[8192,8303],[8304,8351],[8352,8399],[8400,8447],[8448,8527],[8528,8591],[8592,8703],[8704,8959],[8960,9215],[9216,9279],[9280,9311],[9312,9471],[9472,9599],[9600,9631],[9632,9727],[9728,9983],[9984,10175],[10240,10495],[11904,12031],[12032,12255],[12272,12287],[12288,12351],[12352,12447],[12448,12543],[12544,12591],[12592,12687],[12688,12703],[12704,12735],[12800,13055],[13056,13311],[13312,19893],[19968,40959],[40960,42127],[42128,42191],[44032,55203],[55296,56191],[56192,56319],[56320,57343],[57344,63743],[63744,64255],[64256,64335],[64336,65023],[65056,65071],[65072,65103],[65104,65135],[65136,65278],[65279,65279],[65280,65519],[65520,65533]],u={minimumChars:8,strengthScaleFactor:1,bannedPasswords:window.COMPLEXIFY_BANLIST||[],banMode:"strict"};return t.isFunction(e)&&!n&&(n=e,e={}),e=t.extend(u,e),this.each(function(){t(this).val()&&o.apply(this)}),this.each(function(){t(this).bind("keyup focus input propertychange mouseup",o)})}})}(t)}).call(e,n(0))},232:function(t,e,n){(function(t,e){+function(t){"use strict";function e(){var t=document.createElement("bootstrap"),e={WebkitTransition:"webkitTransitionEnd",MozTransition:"transitionend",OTransition:"oTransitionEnd otransitionend",transition:"transitionend"};for(var n in e)if(void 0!==t.style[n])return{end:e[n]}}t.fn.emulateTransitionEnd=function(e){var n=!1,i=this;t(this).one(t.support.transition.end,function(){n=!0});var r=function(){n||t(i).trigger(t.support.transition.end)};return setTimeout(r,e),this},t(function(){t.support.transition=e()})}(t),function(t){"use strict";var e=function(e){this.element=t(e)};e.prototype.show=function(){var e=this.element,n=e.closest("ul:not(.dropdown-menu)"),i=e.attr("data-target");if(i||(i=e.attr("href"),i=i&&i.replace(/.*(?=#[^\s]*$)/,"")),!e.parent("li").hasClass("active")){var r=n.find(".active:last a")[0],o=t.Event("show.bs.tab",{relatedTarget:r});if(e.trigger(o),!o.isDefaultPrevented()){var s=t(i);this.activate(e.parent("li"),n),this.activate(s,s.parent(),function(){e.trigger({type:"shown.bs.tab",relatedTarget:r})})}}},e.prototype.activate=function(e,n,i){function r(){o.removeClass("active").find("> .dropdown-menu > .active").removeClass("active"),e.addClass("active"),s?(e[0].offsetWidth,e.addClass("in")):e.removeClass("fade"),e.parent(".dropdown-menu")&&e.closest("li.dropdown").addClass("active"),i&&i()}var o=n.find("> .active"),s=i&&t.support.transition&&o.hasClass("fade");s?o.one(t.support.transition.end,r).emulateTransitionEnd(150):r(),o.removeClass("in")};var n=t.fn.tab;t.fn.tab=function(n){return this.each(function(){var i=t(this),r=i.data("bs.tab");r||i.data("bs.tab",r=new e(this)),"string"==typeof n&&r[n]()})},t.fn.tab.Constructor=e,t.fn.tab.noConflict=function(){return t.fn.tab=n,this},t(document).on("click.bs.tab.data-api",'[data-toggle="tab"], [data-toggle="pill"]',function(e){e.preventDefault(),t(this).tab("show")})}(t),function(t){"use strict";var e=function(t,e){this.type=this.options=this.enabled=this.timeout=this.hoverState=this.$element=null,this.init("tooltip",t,e)};e.DEFAULTS={animation:!0,placement:"top",selector:!1,template:'
',trigger:"hover focus",title:"",delay:0,html:!1,container:"body"},e.prototype.init=function(e,n,i){this.enabled=!0,this.type=e,this.$element=t(n),this.options=this.getOptions(i);for(var r=this.options.trigger.split(" "),o=r.length;o--;){var s=r[o];if("click"==s)this.$element.on("click."+this.type,this.options.selector,t.proxy(this.toggle,this));else if("manual"!=s){var a="hover"==s?"mouseenter":"focus",l="hover"==s?"mouseleave":"blur";this.$element.on(a+"."+this.type,this.options.selector,t.proxy(this.enter,this)),this.$element.on(l+"."+this.type,this.options.selector,t.proxy(this.leave,this))}}this.options.selector?this._options=t.extend({},this.options,{trigger:"manual",selector:""}):this.fixTitle()},e.prototype.getDefaults=function(){return e.DEFAULTS},e.prototype.getOptions=function(e){return e=t.extend({},this.getDefaults(),this.$element.data(),e),e.delay&&"number"==typeof e.delay&&(e.delay={show:e.delay,hide:e.delay}),e},e.prototype.getDelegateOptions=function(){var e={},n=this.getDefaults();return this._options&&t.each(this._options,function(t,i){n[t]!=i&&(e[t]=i)}),e},e.prototype.enter=function(e){var n=e instanceof this.constructor?e:t(e.currentTarget)[this.type](this.getDelegateOptions()).data("bs."+this.type);if(clearTimeout(n.timeout),n.hoverState="in",!n.options.delay||!n.options.delay.show)return n.show();n.timeout=setTimeout(function(){"in"==n.hoverState&&n.show()},n.options.delay.show)},e.prototype.leave=function(e){var n=e instanceof this.constructor?e:t(e.currentTarget)[this.type](this.getDelegateOptions()).data("bs."+this.type);if(clearTimeout(n.timeout),n.hoverState="out",!n.options.delay||!n.options.delay.hide)return n.hide();n.timeout=setTimeout(function(){"out"==n.hoverState&&n.hide()},n.options.delay.hide)},e.prototype.show=function(){var e=t.Event("show.bs."+this.type);if(this.hasContent()&&this.enabled){if(this.$element.trigger(e),e.isDefaultPrevented())return;var n=this.tip();this.setContent(),this.options.animation&&n.addClass("fade");var i="function"==typeof this.options.placement?this.options.placement.call(this,n[0],this.$element[0]):this.options.placement,r=/\s?auto?\s?/i,o=r.test(i);o&&(i=i.replace(r,"")||"top"),n.detach().css({top:0,left:0,display:"block"}).addClass(i),this.options.container?n.appendTo(this.options.container):n.insertAfter(this.$element);var s=this.getPosition(),a=n[0].offsetWidth,l=n[0].offsetHeight;if(o){var u=this.$element.parent(),c=i,h=document.documentElement.scrollTop||document.body.scrollTop,d="body"==this.options.container?window.innerWidth:u.outerWidth(),p="body"==this.options.container?window.innerHeight:u.outerHeight(),f="body"==this.options.container?0:u.offset().left;i="bottom"==i&&s.top+s.height+l-h>p?"top":"top"==i&&s.top-h-l<0?"bottom":"right"==i&&s.right+a>d?"left":"left"==i&&s.left-a').insertAfter(t(this)).on("click",e),o.trigger(i=t.Event("show.bs.dropdown")),i.isDefaultPrevented())return;o.toggleClass("open").trigger("shown.bs.dropdown"),r.focus()}return!1}},o.prototype.keydown=function(e){if(/(38|40|27)/.test(e.keyCode)){var i=t(this);if(e.preventDefault(),e.stopPropagation(),!i.is(".disabled, :disabled")){var o=n(i),s=o.hasClass("open");if(!s||s&&27==e.keyCode)return 27==e.which&&o.find(r).focus(),i.click();var a=t("[role=menu] li:not(.divider):visible a",o);if(a.length){var l=a.index(a.filter(":focus"));38==e.keyCode&&l>0&&l--,40==e.keyCode&&l

'}),e.prototype=t.extend({},t.fn.tooltip.Constructor.prototype),e.prototype.constructor=e,e.prototype.getDefaults=function(){return e.DEFAULTS},e.prototype.setContent=function(){var t=this.tip(),e=this.getTitle(),n=this.getContent();t.find(".popover-title")[this.options.html?"html":"text"](e),t.find(".popover-content")[this.options.html?"html":"text"](n),t.removeClass("fade top bottom left right in"),t.find(".popover-title").html()||t.find(".popover-title").hide()},e.prototype.hasContent=function(){return this.getTitle()||this.getContent()},e.prototype.getContent=function(){var t=this.$element,e=this.options;return t.attr("data-content")||("function"==typeof e.content?e.content.call(t[0]):e.content)},e.prototype.arrow=function(){return this.$arrow=this.$arrow||this.tip().find(".arrow")},e.prototype.tip=function(){return this.$tip||(this.$tip=t(this.options.template)),this.$tip};var n=t.fn.popover;t.fn.popover=function(n){return this.each(function(){var i=t(this),r=i.data("bs.popover"),o="object"==typeof n&&n;r||i.data("bs.popover",r=new e(this,o)),"string"==typeof n&&r[n]()})},t.fn.popover.Constructor=e,t.fn.popover.noConflict=function(){return t.fn.popover=n,this}}(e)}).call(e,n(0),n(0))},233:function(t,e){t.exports=function(){throw new Error("define cannot be used indirect")}},234:function(t,e,n){"use strict";(function(t,e,i,r,o){function s(t){return t&&t.__esModule?t:{default:t}}function a(e,n,i){function r(t){var e=o(t),n={placeholder:"Click to select",closeOnSelect:!e.is("[MULTIPLE]"),dropdownAutoWidth:!0,containerCssClass:"select2-minwidth"};return t.select2(n)}t.fn.select2&&(void 0===e&&(e=20),void 0===n&&(n=3e3),i=i||o("select"),i.each(function(){var t=o(this).not("[multiple]"),i=t.find("option").length;in||t.hasClass("no-autocomplete")||r(t)}))}function l(){o("select[refresh_on_change='true']").off("change").change(function(){var t=o(this),e=t.val(),n=t.attr("refresh_on_change_values");if(n){n=n.split(",");var i=t.attr("last_selected_value");if(-1===o.inArray(e,n)&&-1===o.inArray(i,n))return}o(window).trigger("refresh_on_change"),o(document).trigger("convert_to_values"),t.get(0).form.submit()}),o(":checkbox[refresh_on_change='true']").off("click").click(function(){var t=o(this),e=t.val(),n=t.attr("refresh_on_change_values");if(n){n=n.split(",");var i=t.attr("last_selected_value");if(-1===o.inArray(e,n)&&-1===o.inArray(i,n))return}o(window).trigger("refresh_on_change"),t.get(0).form.submit()}),o("a[confirm]").off("click").click(function(){return confirm(o(this).attr("confirm"))})}var u=n(176),c=s(u),h=n(202),d=s(h),p=n(203),f=s(p),g=n(235),v=s(g),m=n(201),y=s(m),b=n(204),w=s(b),_=n(61);s(_);window.$=t,window._=i,window.Backbone=r,window.panels=c.default,i.extend(window,d.default),window.async_save_text=f.default,window.make_popupmenu=v.default.make_popupmenu,window.make_popup_menus=v.default.make_popup_menus,window.init_tag_click_function=y.default,window.init_refresh_on_change=l,o(document).ready(function(){function t(){void 0!==Galaxy.root?o.getJSON(Galaxy.root+"api/webhooks/onload/all",function(t){i.each(t,function(t){t.activate&&t.script&&(o("