From 15bde671e772a0ec508cf614f688d1c3e255a419 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Fri, 15 Jan 2021 19:49:45 -0500 Subject: [PATCH] Fix certain classes of workflow extraction on copied objects. --- lib/galaxy/workflow/extract.py | 37 +++++++++++++++++++-- test/unit/workflows/test_extract_summary.py | 1 + 2 files changed, 35 insertions(+), 3 deletions(-) diff --git a/lib/galaxy/workflow/extract.py b/lib/galaxy/workflow/extract.py index cf0f4a96b7b..a6c3354513d 100644 --- a/lib/galaxy/workflow/extract.py +++ b/lib/galaxy/workflow/extract.py @@ -129,12 +129,14 @@ def extract_steps(trans, history=None, job_ids=None, dataset_ids=None, dataset_c assoc_name = assoc.name if ToolOutputCollectionPart.is_named_collection_part_name(assoc_name): continue + if assoc_name.startswith("__new_primary_file"): + continue if job in summary.implicit_map_jobs: hid = None for implicit_pair in jobs[job]: query_assoc_name, dataset_collection = implicit_pair if query_assoc_name == assoc_name or assoc_name.startswith("__new_primary_file_%s|" % query_assoc_name): - hid = dataset_collection.hid + hid = summary.hid(dataset_collection) if hid is None: template = "Failed to find matching implicit job - job id is %s, implicit pairs are %s, assoc_name is %s." message = template % (job.id, jobs[job], assoc_name) @@ -142,9 +144,12 @@ def extract_steps(trans, history=None, job_ids=None, dataset_ids=None, dataset_c raise Exception("Failed to extract job.") else: if hasattr(assoc, "dataset"): - hid = assoc.dataset.hid + has_hid = assoc.dataset else: - hid = assoc.dataset_collection_instance.hid + has_hid = assoc.dataset_collection_instance + hid = summary.hid(has_hid) + if hid in hid_to_output_pair: + log.warning("duplicate hid found in extract_steps [%s]" % hid) hid_to_output_pair[hid] = (step, assoc.name) return steps @@ -196,8 +201,29 @@ class WorkflowSummary: self.implicit_map_jobs = [] self.collection_types = {} + self.hda_hid_in_history = {} + self.hdca_hid_in_history = {} + self.__summarize() + def hid(self, object): + if object.history_content_type == "dataset_collection": + if object.id in self.hdca_hid_in_history: + return self.hdca_hid_in_history[object.id] + elif object.history == self.history: + return object.hid + else: + log.warning("extraction issue, using hdca hid from outside current history and unmapped") + return object.hid + else: + if object.id in self.hda_hid_in_history: + return self.hda_hid_in_history[object.id] + elif object.history == self.history: + return object.hid + else: + log.warning("extraction issue, using hda hid from outside current history and unmapped") + return object.hid + def __summarize(self): # Make a first pass handle all singleton jobs, input dataset and dataset collections # just grab the implicitly mapped jobs and handle in second pass. Second pass is @@ -215,7 +241,10 @@ class WorkflowSummary: self.__summarize_dataset(content) def __summarize_dataset_collection(self, dataset_collection): + hid_in_history = dataset_collection.hid dataset_collection = self.__original_hdca(dataset_collection) + self.hdca_hid_in_history[dataset_collection.id] = hid_in_history + hid = dataset_collection.hid self.collection_types[hid] = dataset_collection.collection.collection_type cja = dataset_collection.creating_job_associations @@ -264,7 +293,9 @@ class WorkflowSummary: if not self.__check_state(dataset): return + hid_in_history = dataset.hid original_hda = self.__original_hda(dataset) + self.hda_hid_in_history[original_hda.id] = hid_in_history if not original_hda.creating_job_associations: self.jobs[FakeJob(dataset)] = [(None, dataset)] diff --git a/test/unit/workflows/test_extract_summary.py b/test/unit/workflows/test_extract_summary.py index 4b437269421..18be155c675 100644 --- a/test/unit/workflows/test_extract_summary.py +++ b/test/unit/workflows/test_extract_summary.py @@ -110,6 +110,7 @@ class MockTrans: class MockHda: def __init__(self, state='ok', output_name='out1', job=None): + self.hid = 1 self.id = 123 self.state = state self.copied_from_history_dataset_association = None