Merge pull request #5013 from jmchilton/workflow_and_collection_state

Improved Collection and Workflow State with Applications
This commit is contained in:
Martin Cech
2017-11-30 10:16:23 -05:00
committed by GitHub
72 changed files with 2914 additions and 594 deletions
+94 -8
View File
@@ -16,10 +16,14 @@ var HDCAListItemView = _super.extend(
/** event listeners */
_setUpListeners: function() {
_super.prototype._setUpListeners.call(this);
var renderListen = (model, options) => {
this.render();
};
if (this.model.jobStatesSummary) {
this.listenTo(this.model.jobStatesSummary, "change", renderListen);
}
this.listenTo(this.model, {
"change:tags change:populated change:visible": function(model, options) {
this.render();
}
"change:tags change:visible change:state": renderListen
});
},
@@ -43,13 +47,95 @@ var HDCAListItemView = _super.extend(
_swapNewRender: function($newRender) {
_super.prototype._swapNewRender.call(this, $newRender);
//TODO: model currently has no state
var state = !this.model.get("populated") ? STATES.RUNNING : STATES.OK;
//if( this.model.has( 'state' ) ){
var state;
var jobStatesSummary = this.model.jobStatesSummary;
if (jobStatesSummary) {
if (jobStatesSummary.new()) {
state = "loading";
} else if (jobStatesSummary.errored()) {
state = "error";
} else if (jobStatesSummary.terminal()) {
state = "ok";
} else if (jobStatesSummary.running()) {
state = "running";
} else {
state = "queued";
}
} else if (this.model.get("job_source_id")) {
// Initial rendering - polling will fill in more details in a bit.
state = "loading";
} else {
state = this.model.get("populated_state") ? STATES.OK : STATES.RUNNING;
}
this.$el.addClass(`state-${state}`);
//}
var stateDescription = this.stateDescription();
this.$(".state-description").html(stateDescription);
return this.$el;
},
stateDescription: function() {
var collection = this.model;
var elementCount = collection.get("element_count");
var jobStateSource = collection.get("job_source_type");
var collectionType = this.model.get("collection_type");
var collectionTypeDescription;
if (collectionType == "list") {
collectionTypeDescription = "list";
} else if (collectionType == "paired") {
collectionTypeDescription = "dataset pair";
} else if (collectionType == "list:paired") {
collectionTypeDescription = "list of pairs";
} else {
collectionTypeDescription = "nested list";
}
var itemsDescription = "";
if (elementCount == 1) {
itemsDescription = ` with 1 item`;
} else if (elementCount) {
itemsDescription = ` with ${elementCount} items`;
}
var jobStatesSummary = collection.jobStatesSummary;
var simpleDescription = `${collectionTypeDescription}${itemsDescription}`;
if (!jobStateSource || jobStateSource == "Job") {
return `a ${simpleDescription}`;
} else if (!jobStatesSummary || !jobStatesSummary.hasDetails()) {
return `
<div class="progress state-progress">
<span class="note">Loading job data for ${collectionTypeDescription}.<span class="blinking">..</span></span>
<div class="progress-bar info" style="width:100%">
</div>`;
} else {
var isNew = jobStatesSummary.new();
var jobCount = isNew ? null : jobStatesSummary.jobCount();
if (isNew) {
return `
<div class="progress state-progress">
<span class="note">Creating jobs.<span class="blinking">..</span></span>
<div class="progress-bar info" style="width:100%">
</div>`;
} else if (jobStatesSummary.errored()) {
var errorCount = jobStatesSummary.numInError();
return `a ${collectionTypeDescription} with ${errorCount} / ${jobCount} jobs in error`;
} else if (jobStatesSummary.terminal()) {
return `a ${simpleDescription}`;
} else {
var running = jobStatesSummary.states()["running"] || 0;
var ok = jobStatesSummary.states()["ok"] || 0;
var okPercent = ok / (jobCount * 1.0);
var runningPercent = running / (jobCount * 1.0);
var otherPercent = 1.0 - okPercent - runningPercent;
var jobsStr = jobCount && jobCount > 1 ? `${jobCount} jobs` : `a job`;
return `
<div class="progress state-progress">
<span class="note">${jobsStr} generating a ${collectionTypeDescription}</span>
<div class="progress-bar ok" style="width:${okPercent * 100.0}%"></div>
<div class="progress-bar running" style="width:${runningPercent * 100.0}%"></div>
<div class="progress-bar new" style="width:${otherPercent * 100.0}%">
</div>`;
}
}
},
// ......................................................................... misc
/** String representation */
toString: function() {
@@ -69,7 +155,6 @@ HDCAListItemView.prototype.templates = (() => {
}
});
// could steal this from hda-base (or use mixed content)
var titleBarTemplate = collection => `
<div class="title-bar clear" tabindex="0">
<span class="state-icon"></span>
@@ -77,7 +162,8 @@ HDCAListItemView.prototype.templates = (() => {
<span class="hid">${collection.hid}</span>
<span class="name">${_.escape(collection.name)}</span>
</div>
<div class="subtitle"></div>
<div class="state-description">
</div>
${HISTORY_ITEM_LI.nametagTemplate(collection)}
</div>
`;
@@ -2,6 +2,7 @@ import CONTROLLED_FETCH_COLLECTION from "mvc/base/controlled-fetch-collection";
import HDA_MODEL from "mvc/history/hda-model";
import HDCA_MODEL from "mvc/history/hdca-model";
import HISTORY_PREFS from "mvc/history/history-preferences";
import JOB_STATES_MODEL from "mvc/history/job-states-model";
import BASE_MVC from "mvc/base-mvc";
import AJAX_QUEUE from "utils/ajax-queue";
@@ -37,6 +38,10 @@ var HistoryContents = _super.extend(BASE_MVC.LoggableMixin).extend({
/** Set up */
initialize: function(models, options) {
this.on({
"sync add": this.trackJobStates
});
options = options || {};
_super.prototype.initialize.call(this, models, options);
@@ -53,6 +58,29 @@ var HistoryContents = _super.extend(BASE_MVC.LoggableMixin).extend({
this.model.prototype.idAttribute = "type_id";
},
trackJobStates: function() {
this.each(historyContent => {
if (historyContent.has("job_states_summary")) {
return;
}
if (historyContent.attributes.history_content_type === "dataset_collection") {
var jobSourceType = historyContent.attributes.job_source_type;
var jobSourceId = historyContent.attributes.job_source_id;
if (jobSourceType) {
this.jobStateSummariesCollection.add({
id: jobSourceId,
model: jobSourceType,
history_id: this.history_id,
collection_id: historyContent.attributes.id
});
var jobStatesSummary = this.jobStateSummariesCollection.get(jobSourceId);
historyContent.jobStatesSummary = jobStatesSummary;
}
}
});
},
// ........................................................................ composite collection
/** since history content is a mix, override model fn into a factory, creating based on history_content_type */
model: function(attrs, options) {
@@ -82,17 +110,30 @@ var HistoryContents = _super.extend(BASE_MVC.LoggableMixin).extend({
};
},
stopPolling: function() {
if (this.jobStateSummariesCollection) {
this.jobStateSummariesCollection.active = false;
this.jobStateSummariesCollection.clearUpdateTimeout();
}
},
setHistoryId: function(newId) {
this.stopPolling();
this.historyId = newId;
this._setUpWebStorage();
if (newId) {
// If actually reflecting a history - setup storage and monitor jobs.
this._setUpWebStorage();
this.jobStateSummariesCollection = new JOB_STATES_MODEL.JobStatesSummaryCollection();
this.jobStateSummariesCollection.historyId = newId;
this.jobStateSummariesCollection.monitor();
}
},
/** Set up client side storage. Currently PersistanStorage keyed under 'history:<id>' */
_setUpWebStorage: function(initialSettings) {
// TODO: use initialSettings
if (!this.historyId) {
return;
}
this.storage = new HISTORY_PREFS.HistoryPrefs({
id: HISTORY_PREFS.HistoryPrefs.historyStorageKey(this.historyId)
});
@@ -303,17 +344,6 @@ var HistoryContents = _super.extend(BASE_MVC.LoggableMixin).extend({
return this.fetch(options);
},
/** specialty fetch method for retrieving the element_counts of all hdcas in the history */
fetchCollectionCounts: function(options) {
options = options || {};
options.keys = ["type_id", "element_count"].join(",");
options.filters = _.extend(options.filters || {}, {
history_content_type: "dataset_collection"
});
options.remove = false;
return this.fetch(options);
},
// ............. quasi-batch ops
// TODO: to batch
/** helper that fetches using filterParams then calls save on each fetched using updateWhat as the save params */
@@ -237,6 +237,13 @@ var History = Backbone.Model.extend(BASE_MVC.LoggableMixin).extend(
}
},
stopPolling: function() {
this.clearUpdateTimeout();
if (this.contents) {
this.contents.stopPolling();
}
},
// ........................................................................ ajax
/** override to use actual Dates objects for create/update times */
parse: function(response, options) {
@@ -47,9 +47,6 @@ var HistoryView = _super.extend(
/** string used for search placeholder */
searchPlaceholder: _l("search datasets"),
/** @type {Number} ms to wait after history load to fetch/decorate hdcas with element_count */
FETCH_COLLECTION_COUNTS_DELAY: 2000,
// ......................................................................... SET UP
/** Set up the view, bind listeners.
* @param {Object} attributes optional settings for the panel
@@ -60,9 +57,6 @@ var HistoryView = _super.extend(
// control contents/behavior based on where (and in what context) the panel is being used
/** where should pages from links be displayed? (default to new tab/window) */
this.linkTarget = attributes.linkTarget || "_blank";
/** timeout id for detailed fetch of collection counts, etc... */
this.detailedFetchTimeoutId = null;
},
/** create and return a collection for when none is initially passed */
@@ -77,20 +71,11 @@ var HistoryView = _super.extend(
freeModel: function() {
_super.prototype.freeModel.call(this);
if (this.model) {
this.model.clearUpdateTimeout();
this.model.stopPolling();
}
this._clearDetailedFetchTimeout();
return this;
},
/** clear the timeout and the cached timeout id */
_clearDetailedFetchTimeout: function() {
if (this.detailedFetchTimeoutId) {
clearTimeout(this.detailedFetchTimeoutId);
this.detailedFetchTimeoutId = null;
}
},
/** create any event listeners for the panel
* @fires: rendered:initial on the first render
* @fires: empty-history when switching to a history with no contents or creating a new history
@@ -101,13 +86,6 @@ var HistoryView = _super.extend(
error: function(model, xhr, options, msg, details) {
this.errorHandler(model, xhr, options, msg, details);
},
"loading-done": () => {
// after the initial load, decorate with more time consuming fields (like HDCA element_counts)
this.detailedFetchTimeoutId = _.delay(() => {
this.detailedFetchTimeoutId = null;
this.model.contents.fetchCollectionCounts();
}, this.FETCH_COLLECTION_COUNTS_DELAY);
},
"views:ready view:attached view:removed": function(view) {
this._renderSelectButton();
},
@@ -378,17 +356,17 @@ var HistoryView = _super.extend(
}),
_clickPrevPage: function(ev) {
this.model.clearUpdateTimeout();
this.model.stopPolling();
this.model.contents.fetchPrevPage();
},
_clickNextPage: function(ev) {
this.model.clearUpdateTimeout();
this.model.stopPolling();
this.model.contents.fetchNextPage();
},
_changePageSelect: function(ev) {
this.model.clearUpdateTimeout();
this.model.stopPolling();
var page = $(ev.currentTarget).val();
this.model.contents.fetchPage(page);
},
@@ -0,0 +1,168 @@
import * as Backbone from "libs/backbone";
import AJAX_QUEUE from "utils/ajax-queue";
/** ms between fetches when checking running jobs/datasets for updates */
var UPDATE_DELAY = 2000;
var NON_TERMINAL_STATES = ["new", "queued", "running"];
var ERROR_STATES = ["error", "deleted"];
/** Fetch state on add or just wait for polling to start. */
var FETCH_STATE_ON_ADD = false;
var BATCH_FETCH_STATE = true;
var JobStatesSummary = Backbone.Model.extend({
url: function() {
return `${Galaxy.root}api/histories/${this.attributes.history_id}/contents/dataset_collections/${
this.attributes.collection_id
}/jobs_summary`;
},
hasDetails: function() {
return this.has("populated_state");
},
new: function() {
return !this.hasDetails() || this.get("populated_state") == "new";
},
errored: function() {
return this.get("populated_state") === "error" || this.anyWithStates(ERROR_STATES);
},
states: function() {
return this.get("states") || {};
},
anyWithState: function(queryState) {
return (this.states()[queryState] || 0) > 0;
},
anyWithStates: function(queryStates) {
var states = this.states();
for (var index in queryStates) {
if ((states[queryStates[index]] || 0) > 0) {
return true;
}
}
return false;
},
numWithStates: function(queryStates) {
var states = this.states();
var count = 0;
for (var index in queryStates) {
count += states[queryStates[index]] || 0;
}
return count;
},
numInError: function() {
return this.numWithStates(ERROR_STATES);
},
running: function() {
return this.anyWithState("running");
},
terminal: function() {
if (this.new()) {
return false;
} else {
var anyNonTerminal = this.anyWithStates(NON_TERMINAL_STATES);
return !anyNonTerminal;
}
},
jobCount: function() {
var states = this.states();
var count = 0;
for (var index in states) {
count += states[index];
}
return count;
},
toString: function() {
return `JobStatesSummary(id=${this.get("id")})`;
}
});
var JobStatesSummaryCollection = Backbone.Collection.extend({
model: JobStatesSummary,
initialize: function() {
if (FETCH_STATE_ON_ADD) {
this.on({
add: model => model.fetch()
});
}
/** cached timeout id for the dataset updater */
this.updateTimeoutId = null;
// this.checkForUpdates();
this.active = true;
},
url: function() {
var nonTerminalModels = this.models.filter(model => {
return !model.terminal();
});
var ids = nonTerminalModels
.map(summary => {
return summary.get("id");
})
.join(",");
var types = nonTerminalModels
.map(summary => {
return summary.get("model");
})
.join(",");
return `${Galaxy.root}api/histories/${this.historyId}/jobs_summary?ids=${ids}&types=${types}`;
},
monitor: function() {
this.clearUpdateTimeout();
if (!this.active) {
return;
}
var _delayThenMonitorAgain = () => {
this.updateTimeoutId = setTimeout(() => {
this.monitor();
}, UPDATE_DELAY);
};
var nonTerminalModels = this.models.filter(model => {
return !model.terminal();
});
if (nonTerminalModels.length > 0 && !BATCH_FETCH_STATE) {
// Allow models to fetch their own details.
var updateFunctions = nonTerminalModels.map(summary => {
return () => {
return summary.fetch();
};
});
return new AJAX_QUEUE.AjaxQueue(updateFunctions).done(_delayThenMonitorAgain);
} else if (nonTerminalModels.length > 0) {
// Batch fetch updated state...
this.fetch({ remove: false }).done(_delayThenMonitorAgain);
} else {
_delayThenMonitorAgain();
}
},
/** clear the timeout and the cached timeout id */
clearUpdateTimeout: function() {
if (this.updateTimeoutId) {
clearTimeout(this.updateTimeoutId);
this.updateTimeoutId = null;
}
},
toString: function() {
return `JobStatesSummaryCollection()`;
}
});
export default { JobStatesSummary, JobStatesSummaryCollection, FETCH_STATE_ON_ADD };
@@ -1,6 +1,7 @@
import _l from "utils/localization";
import HISTORY_MODEL from "mvc/history/history-model";
import HISTORY_VIEW_EDIT from "mvc/history/history-view-edit";
import JOB_STATES_MODEL from "mvc/history/job-states-model";
import historyCopyDialog from "mvc/history/copy-dialog";
import ERROR_MODAL from "mvc/ui/error-modal";
import baseMVC from "mvc/base-mvc";
@@ -733,9 +734,16 @@ var MultiPanelColumns = Backbone.View.extend(baseMVC.LoggableMixin).extend({
this.hdaQueue.add({
name: column.model.id,
fn: function() {
return contents.fetchCurrentPage(fetchOptions).done(() => {
column.panel.renderItems();
});
return contents
.fetchCurrentPage(fetchOptions)
.done(() => {
column.panel.renderItems();
})
.done(() => {
if (!JOB_STATES_MODEL.FETCH_STATE_ON_ADD) {
contents.jobStateSummariesCollection.fetch();
}
});
}
});
// the queue is re-used, so if it's not processing requests - start it again
+31
View File
@@ -1108,6 +1108,37 @@ ul.manage-table-actions li {
margin-left: 0.5em;
}
.state-progress {
border: 1px solid gray;
position: relative;
margin-top: 2px;
margin-bottom: 1px;
.info {
color: @black;
background: @white;
}
.new {
background: @state-default-bg;
}
.running {
background: @state-running-bg;
}
.ok {
background: @state-success-bg;
}
.note {
margin-left: 1em;
position: absolute;
}
}
// State colors
.state-color-new {
+16
View File
@@ -176,6 +176,22 @@
}
}
}
&.state-loading {
background: @state-default-bg;
.state-icon {
.state-icon-running;
}
}
}
.blinking {
animation: blinker 500ms linear infinite;
}
@keyframes blinker {
50% { opacity: 0; }
}
// ---------------------------------------------------------------------------- datasets as list-items
+7
View File
@@ -86,6 +86,13 @@
color: inherit;
}
}
.state-description {
color: #777;
font-size: 90%;
a {
color: inherit;
}
}
}
.primary-actions {
+8
View File
@@ -1131,6 +1131,14 @@ use_interactive = True
# invocation to schedule indefinitely. The default corresponds to 1 month.
#maximum_workflow_invocation_duration = 2678400
# Specify a maximum number of jobs that any given workflow scheduling iteration can create.
# Set this to a positive integer to prevent large collection jobs in a workflow from
# preventing other jobs from executing. This may also mitigate memory issues associated with
# scheduling workflows at the expense of increased total DB traffic because model objects
# are expunged from the SQL alchemy session between workflow invocation scheduling iterations.
# Set to -1 to disable any such maximum (the default).
#maximum_workflow_jobs_per_scheduling_iteration = -1
# Force serial scheduling of workflows within the context of a particular history
#history_local_serial_workflow_scheduling=False
+1
View File
@@ -388,6 +388,7 @@ class Configuration(object):
self.history_local_serial_workflow_scheduling = string_as_bool(kwargs.get('history_local_serial_workflow_scheduling', 'False'))
self.parallelize_workflow_scheduling_within_histories = string_as_bool(kwargs.get('parallelize_workflow_scheduling_within_histories', 'False'))
self.maximum_workflow_invocation_duration = int(kwargs.get("maximum_workflow_invocation_duration", 2678400))
self.maximum_workflow_jobs_per_scheduling_iteration = int(kwargs.get("maximum_workflow_jobs_per_scheduling_iteration", -1))
self.cache_user_job_count = string_as_bool(kwargs.get('cache_user_job_count', False))
self.pbs_application_server = kwargs.get('pbs_application_server', "")
@@ -24,6 +24,7 @@ def set_collection_elements(dataset_collection, type, dataset_instances):
element_index += 1
dataset_collection.elements = elements
dataset_collection.element_count = element_index
return dataset_collection
+10 -2
View File
@@ -44,27 +44,35 @@ class MatchingCollections(object):
self.linked_structure = None
self.unlinked_structures = []
self.collections = {}
self.subcollection_types = {}
def __attempt_add_to_linked_match(self, input_name, hdca, collection_type_description, subcollection_type):
structure = get_structure(hdca, collection_type_description, leaf_subcollection_type=subcollection_type)
if not self.linked_structure:
self.linked_structure = structure
self.collections[input_name] = hdca
self.subcollection_types[input_name] = subcollection_type
else:
if not self.linked_structure.can_match(structure):
raise exceptions.MessageException(CANNOT_MATCH_ERROR_MESSAGE)
self.collections[input_name] = hdca
self.subcollection_types[input_name] = subcollection_type
def slice_collections(self):
return self.linked_structure.walk_collections(self.collections)
def subcollection_mapping_type(self, input_name):
return self.subcollection_types[input_name]
@property
def structure(self):
"""Yield cross product of all unlinked datasets to linked dataset."""
"""Yield cross product of all unlinked collections structures to linked collection structure."""
effective_structure = leaf
for unlinked_structure in self.unlinked_structures:
effective_structure = effective_structure.multiply(unlinked_structure)
linked_structure = self.linked_structure or leaf
linked_structure = self.linked_structure
if linked_structure is None:
linked_structure = leaf
effective_structure = effective_structure.multiply(linked_structure)
return None if effective_structure.is_leaf else effective_structure
+80 -8
View File
@@ -1,12 +1,17 @@
""" Module for reasoning about structure of and matching hierarchical collections of data.
"""
import logging
log = logging.getLogger(__name__)
import six
from .type_description import map_over_collection_type
log = logging.getLogger(__name__)
@six.python_2_unicode_compatible
class Leaf(object):
children_known = True
def __len__(self):
return 1
@@ -18,18 +23,60 @@ class Leaf(object):
def clone(self):
return self
def multiply(self, other_structure):
return other_structure.clone()
def multiply(self, other_structure, uninitialized=False):
if not uninitialized:
return other_structure.clone()
else:
return UnitializedTree(other_structure.collection_type_description)
def sliced_collection_type(self, collection):
return input
def __str__(self):
return "Leaf[]"
leaf = Leaf()
class Tree(object):
class BaseTree(object):
def __init__(self, collection_type_description):
self.collection_type_description = collection_type_description
@six.python_2_unicode_compatible
class UnitializedTree(BaseTree):
children_known = False
def clone(self):
return self
@property
def is_leaf(self):
return False
def __len__(self):
raise Exception("Unknown length")
def multiply(self, other_structure, uninitialized=False):
if other_structure.is_leaf:
return self.clone()
new_collection_type = self.collection_type_description.multiply(other_structure.collection_type_description)
return UnitializedTree(new_collection_type)
def __str__(self):
return "UnitializedTree[collection_type=%s]" % self.collection_type_description
@six.python_2_unicode_compatible
class Tree(BaseTree):
children_known = True
def __init__(self, children, collection_type_description):
super(Tree, self).__init__(collection_type_description)
self.children = children
self.collection_type_description = collection_type_description
@staticmethod
def for_dataset_collection(dataset_collection, collection_type_description):
@@ -107,14 +154,14 @@ class Tree(object):
element_identifiers=element_identifiers,
)
def multiply(self, other_structure):
def multiply(self, other_structure, uninitialized=False):
if other_structure.is_leaf:
return self.clone()
new_collection_type = self.collection_type_description.multiply(other_structure.collection_type_description)
new_children = []
for (identifier, structure) in self.children:
new_children.append((identifier, structure.multiply(other_structure)))
new_children.append((identifier, structure.multiply(other_structure, uninitialized=uninitialized)))
return Tree(new_children, new_collection_type)
@@ -122,6 +169,30 @@ class Tree(object):
cloned_children = [(_[0], _[1].clone()) for _ in self.children]
return Tree(cloned_children, self.collection_type_description)
def __str__(self):
return "Tree[collection_type=%s,children=%s]" % (self.collection_type_description, ",".join(map(lambda identifier_and_element: "%s=%s" % (identifier_and_element[0], identifier_and_element[1]), self.children)))
def tool_output_to_structure(get_sliced_input_collection_type, tool_output, collections_manager):
if not tool_output.collection:
tree = leaf
else:
collection_type_descriptions = collections_manager.collection_type_descriptions
# Okay this is ToolCollectionOutputStructure not a Structure - different
# concepts of structure.
if tool_output.dynamic_structure:
# Two cases collection_type_source and collection_type right?
tree = UnitializedTree(collection_type_descriptions.for_type_description("list")) # list is obviously wrong...
else:
structured_like = tool_output.structure.structured_like
if structured_like:
collection_type = get_sliced_input_collection_type(structured_like)
else:
collection_type = tool_output.structure.collection_type
tree = UnitializedTree(collection_type)
return tree
def dict_map(func, input_dict):
return dict((k, func(v)) for k, v in input_dict.items())
@@ -131,4 +202,5 @@ def get_structure(dataset_collection_instance, collection_type_description, leaf
if leaf_subcollection_type:
collection_type_description = collection_type_description.effective_collection_type_description(leaf_subcollection_type)
return Tree.for_dataset_collection(dataset_collection_instance.collection, collection_type_description)
collection = dataset_collection_instance.collection
return Tree.for_dataset_collection(collection, collection_type_description)
@@ -8,6 +8,7 @@ class CollectionTypeDescriptionFactory(object):
self.type_registry = type_registry
def for_collection_type(self, collection_type):
assert collection_type is not None
return CollectionTypeDescription(collection_type, self)
+1 -1
View File
@@ -1040,7 +1040,7 @@ class JobWrapper(object, HasResourceParameters):
destination_params = job.destination_params
if "__resubmit_delay_seconds" in destination_params:
delay = float(destination_params["__resubmit_delay_seconds"])
if job.seconds_since_update < delay:
if job.seconds_since_updated < delay:
return False
return True
+84 -35
View File
@@ -46,6 +46,40 @@ class DatasetCollectionManager(object):
self.tag_manager = tags.GalaxyTagManager(app.model.context)
self.ldda_manager = lddas.LDDAManager(app)
def precreate_dataset_collection_instance(self, trans, parent, name, implicit_inputs, implicit_output_name, structure):
# TODO: prebuild all required HIDs and send them in so no need to flush in between.
dataset_collection = self.precreate_dataset_collection(structure)
instance = self._create_instance_for_collection(
trans, parent, name, dataset_collection, implicit_inputs=implicit_inputs, implicit_output_name=implicit_output_name, flush=False
)
return instance
def precreate_dataset_collection(self, structure):
if structure.is_leaf or not structure.children_known:
return model.DatasetCollectionElement.UNINITIALIZED_ELEMENT
else:
collection_type_description = structure.collection_type_description
dataset_collection = model.DatasetCollection(populated=False)
dataset_collection.collection_type = collection_type_description.collection_type
elements = []
for index, (identifier, substructure) in enumerate(structure.children):
# TODO: Open question - populate these now or later?
if substructure.is_leaf:
element = model.DatasetCollectionElement.UNINITIALIZED_ELEMENT
else:
element = self.precreate_dataset_collection(substructure)
element = model.DatasetCollectionElement(
element=element,
element_identifier=identifier,
element_index=index,
)
elements.append(element)
dataset_collection.elements = elements
dataset_collection.element_count = len(elements)
return dataset_collection
def create(self, trans, parent, name, collection_type, element_identifiers=None,
elements=None, implicit_collection_info=None, trusted_identifiers=None,
hide_source_items=False, tags=None):
@@ -68,27 +102,30 @@ class DatasetCollectionManager(object):
hide_source_items=hide_source_items,
)
implicit_inputs = []
if implicit_collection_info:
implicit_inputs = implicit_collection_info.get('implicit_inputs', [])
implicit_output_name = None
if implicit_collection_info:
implicit_output_name = implicit_collection_info["implicit_output_name"]
return self._create_instance_for_collection(
trans, parent, name, dataset_collection, implicit_inputs=implicit_inputs, implicit_output_name=implicit_output_name, tags=tags
)
def _create_instance_for_collection(self, trans, parent, name, dataset_collection, implicit_output_name=None, implicit_inputs=None, tags=None, flush=True):
if isinstance(parent, model.History):
dataset_collection_instance = self.model.HistoryDatasetCollectionAssociation(
collection=dataset_collection,
name=name,
)
if implicit_collection_info:
for input_name, input_collection in implicit_collection_info["implicit_inputs"]:
if implicit_inputs:
for input_name, input_collection in implicit_inputs:
dataset_collection_instance.add_implicit_input_collection(input_name, input_collection)
for output_dataset in implicit_collection_info.get("outputs"):
if output_dataset not in trans.sa_session:
output_dataset = trans.sa_session.query(type(output_dataset)).get(output_dataset.id)
if isinstance(output_dataset, model.HistoryDatasetAssociation):
output_dataset.hidden_beneath_collection_instance = dataset_collection_instance
elif isinstance(output_dataset, model.HistoryDatasetCollectionAssociation):
dataset_collection_instance.add_implicit_input_collection(input_name, input_collection)
else:
# dataset collection, don't need to do anything...
pass
trans.sa_session.add(output_dataset)
dataset_collection_instance.implicit_output_name = implicit_collection_info["implicit_output_name"]
if implicit_output_name:
dataset_collection_instance.implicit_output_name = implicit_output_name
log.debug("Created collection with %d elements" % (len(dataset_collection_instance.collection.elements)))
# Handle setting hid
@@ -105,37 +142,26 @@ class DatasetCollectionManager(object):
message = "Internal logic error - create called with unknown parent type %s" % type(parent)
log.exception(message)
raise MessageException(message)
tags = tags or {}
if implicit_collection_info:
for _, v in implicit_collection_info.get('implicit_inputs', []):
for tag in [t for t in v.tags if t.user_tname == 'name']:
tags[tag.value] = tag
for _, tag in tags.items():
dataset_collection_instance.tags.append(tag.copy(cls=model.HistoryDatasetCollectionTagAssociation))
return self.__persist(dataset_collection_instance)
tags = self._append_tags(dataset_collection_instance, implicit_inputs, tags)
return self.__persist(dataset_collection_instance, flush=flush)
def create_dataset_collection(self, trans, collection_type, element_identifiers=None, elements=None,
hide_source_items=None):
# Make sure at least one of these is None.
assert element_identifiers is None or elements is None
if element_identifiers is None and elements is None:
raise RequestParameterInvalidException(ERROR_INVALID_ELEMENTS_SPECIFICATION)
if not collection_type:
raise RequestParameterInvalidException(ERROR_NO_COLLECTION_TYPE)
collection_type_description = self.collection_type_descriptions.for_collection_type(collection_type)
# If we have elements, this is an internal request, don't need to load
# objects from identifiers.
if elements is None:
if collection_type_description.has_subcollections():
# Nested collection - recursively create collections and update identifiers.
self.__recursively_create_collections(trans, element_identifiers)
new_collection = False
for element_identifier in element_identifiers:
if element_identifier.get("src") == "new_collection" and element_identifier.get('collection_type') == '':
new_collection = True
elements = self.__load_elements(trans, element_identifier['element_identifiers'])
if not new_collection:
elements = self.__load_elements(trans, element_identifiers)
elements = self._element_identifiers_to_elements(trans, collection_type_description, element_identifiers)
# else if elements is set, it better be an ordered dict!
if elements is not self.ELEMENTS_UNINITIALIZED:
@@ -150,6 +176,28 @@ class DatasetCollectionManager(object):
dataset_collection.collection_type = collection_type
return dataset_collection
def _element_identifiers_to_elements(self, trans, collection_type_description, element_identifiers):
if collection_type_description.has_subcollections():
# Nested collection - recursively create collections and update identifiers.
self.__recursively_create_collections(trans, element_identifiers)
new_collection = False
for element_identifier in element_identifiers:
if element_identifier.get("src") == "new_collection" and element_identifier.get('collection_type') == '':
new_collection = True
elements = self.__load_elements(trans, element_identifier['element_identifiers'])
if not new_collection:
elements = self.__load_elements(trans, element_identifiers)
return elements
def _append_tags(self, dataset_collection_instance, implicit_inputs=None, tags=None):
tags = tags or {}
implicit_inputs = implicit_inputs or []
for _, v in implicit_inputs:
for tag in [t for t in v.tags if t.user_tname == 'name']:
tags[tag.value] = tag
for _, tag in tags.items():
dataset_collection_instance.tags.append(tag.copy(cls=model.HistoryDatasetCollectionTagAssociation))
def set_collection_elements(self, dataset_collection, dataset_instances):
if dataset_collection.populated:
raise Exception("Cannot reset elements of an already populated dataset collection.")
@@ -240,10 +288,11 @@ class DatasetCollectionManager(object):
collections = list(filter(query.direct_match, collections))
return collections
def __persist(self, dataset_collection_instance):
def __persist(self, dataset_collection_instance, flush=True):
context = self.model.context
context.add(dataset_collection_instance)
context.flush()
if flush:
context.flush()
return dataset_collection_instance
def __recursively_create_collections(self, trans, element_identifiers):
+11 -6
View File
@@ -122,12 +122,17 @@ def dictify_dataset_collection_instance(dataset_collection_instance, parent, sec
def dictify_element(element):
dictified = element.to_dict(view="element")
object_detials = element.element_object.to_dict()
if element.child_collection:
# Recursively yield elements for each nested collection...
child_collection = element.child_collection
object_detials["elements"] = [dictify_element(_) for _ in child_collection.elements]
object_detials["populated"] = child_collection.populated
element_object = element.element_object
if element_object is not None:
object_detials = element.element_object.to_dict()
if element.child_collection:
# Recursively yield elements for each nested collection...
child_collection = element.child_collection
object_detials["elements"] = [dictify_element(_) for _ in child_collection.elements]
object_detials["populated"] = child_collection.populated
object_detials["element_count"] = child_collection.element_count
else:
object_detials = None
dictified["object"] = object_detials
return dictified
+13 -18
View File
@@ -117,12 +117,13 @@ class DCSerializer(base.ModelSerializer):
'create_time',
'update_time',
'collection_type',
'populated',
'populated_state',
'populated_state_message',
'element_count',
])
self.add_view('detailed', [
'elements'
'populated',
'elements',
], include_keys_from='summary')
def add_serializers(self):
@@ -130,7 +131,6 @@ class DCSerializer(base.ModelSerializer):
self.serializers.update({
'model_class' : lambda *a, **c: 'DatasetCollection',
'elements' : self.serialize_elements,
'element_count' : self.serialize_element_count
})
def serialize_elements(self, item, key, **context):
@@ -140,14 +140,6 @@ class DCSerializer(base.ModelSerializer):
returned.append(serialized)
return returned
def serialize_element_count(self, item, key, **context):
"""Return the count of elements for this collection."""
# TODO: app.model.context -> session
# TODO: to the container interface (dataset_collection_contents)
return (self.app.model.context.query(model.DatasetCollectionElement)
.filter(model.DatasetCollectionElement.dataset_collection_id == item.id)
.count())
class DCASerializer(base.ModelSerializer):
"""
@@ -163,12 +155,13 @@ class DCASerializer(base.ModelSerializer):
'id',
'create_time', 'update_time',
'collection_type',
'populated',
'populated_state',
'populated_state_message',
'element_count',
])
self.add_view('detailed', [
'elements'
'populated',
'elements',
], include_keys_from='summary')
def add_serializers(self):
@@ -184,7 +177,7 @@ class DCASerializer(base.ModelSerializer):
'populated_state',
'populated_state_message',
'elements',
'element_count'
'element_count',
]
for key in collection_keys:
self.serializers[key] = self._proxy_to_dataset_collection(key=key)
@@ -221,15 +214,15 @@ class HDCASerializer(
'history_content_type',
'collection_type',
'populated',
'populated_state',
'populated_state_message',
'element_count',
'job_source_id',
'job_source_type',
'name',
'type_id',
'history_id',
'hid',
'history_content_type',
'deleted',
# 'purged',
'visible',
@@ -238,6 +231,7 @@ class HDCASerializer(
'tags', # TODO: detail view only (maybe)
])
self.add_view('detailed', [
'populated',
'elements'
], include_keys_from='summary')
@@ -254,6 +248,7 @@ class HDCASerializer(
'history_id' : self.serialize_id,
'history_content_type' : lambda *a, **c: self.hdca_manager.model_class.content_type,
'type_id' : self.serialize_type_id,
'job_source_id' : self.serialize_id,
'url' : lambda i, k, **c: self.url_for('history_content_typed',
history_id=self.app.security.encode_id(i.history_id),
+75 -1
View File
@@ -3,10 +3,12 @@ import logging
from boltons.iterutils import remap
from six import string_types
from sqlalchemy import and_, false, or_
from sqlalchemy import and_, false, func, or_
from sqlalchemy.orm import aliased
from sqlalchemy.sql import select
from galaxy import model
from galaxy.exceptions import RequestParameterInvalidException
from galaxy.managers.collections import DatasetCollectionManager
from galaxy.managers.hdas import HDAManager
from galaxy.managers.lddas import LDDAManager
@@ -243,3 +245,75 @@ class JobSearch(object):
log.info("Searching jobs finished %s", search_timer)
return job
return None
def fetch_job_states(app, sa_session, job_source_ids, job_source_types):
decode = app.security.decode_id
assert len(job_source_ids) == len(job_source_types)
job_ids = set()
implicit_collection_job_ids = set()
for job_source_id, job_source_type in zip(job_source_ids, job_source_types):
if job_source_type == "Job":
job_ids.add(job_source_id)
elif job_source_type == "ImplicitCollectionJobs":
implicit_collection_job_ids.add(job_source_id)
else:
raise RequestParameterInvalidException("Invalid job source type %s found." % job_source_type)
# TODO: use above sets and optimize queries on second pass.
rval = []
for job_source_id, job_source_type in zip(job_source_ids, job_source_types):
if job_source_type == "Job":
rval.append(summarize_jobs_to_dict(sa_session, sa_session.query(model.Job).get(decode(job_source_id))))
else:
rval.append(summarize_jobs_to_dict(sa_session, sa_session.query(model.ImplicitCollectionJobs).get(decode(job_source_id))))
return rval
def summarize_jobs_to_dict(sa_session, jobs_source):
"""Proudce a summary of jobs for job summary endpoints.
:type jobs_source: a Job or ImplicitCollectionJobs or None
:param jobs_source: the object to summarize
:rtype: dict
:returns: dictionary containing job summary information
"""
rval = None
if jobs_source is None:
pass
elif isinstance(jobs_source, model.Job):
rval = {
"populated_state": "ok",
"states": {jobs_source.state: 1},
"model": "Job",
"id": jobs_source.id,
}
else:
populated_state = jobs_source.populated_state
rval = {
"id": jobs_source.id,
"populated_state": populated_state,
"model": "ImplicitCollectionJobs",
}
if populated_state == "ok":
# produce state summary...
states = {}
join = model.ImplicitCollectionJobs.table.join(
model.ImplicitCollectionJobsJobAssociation.table.join(model.Job)
)
statement = select(
[model.Job.state, func.count("*")]
).select_from(
join
).where(
model.ImplicitCollectionJobs.id == jobs_source.id
).group_by(
model.Job.state
)
for row in sa_session.execute(statement):
states[row[0]] = row[1]
rval["states"] = states
return rval
+222 -49
View File
@@ -117,6 +117,19 @@ class HasName:
return name
class UsesCreateAndUpdateTime:
@property
def seconds_since_updated(self):
update_time = self.update_time or galaxy.model.orm.now.now() # In case not yet flushed
return (galaxy.model.orm.now.now() - update_time).total_seconds()
@property
def seconds_since_created(self):
create_time = self.create_time or galaxy.model.orm.now.now() # In case not yet flushed
return (galaxy.model.orm.now.now() - create_time).total_seconds()
class JobLike:
def _init_metrics(self):
@@ -424,7 +437,7 @@ class TaskMetricNumeric(BaseJobMetric):
pass
class Job(object, JobLike, Dictifiable):
class Job(object, JobLike, UsesCreateAndUpdateTime, Dictifiable):
dict_collection_visible_keys = ['id', 'state', 'exit_code', 'update_time', 'create_time']
dict_element_visible_keys = ['id', 'state', 'exit_code', 'update_time', 'create_time']
@@ -805,10 +818,6 @@ class Job(object, JobLike, Dictifiable):
config_value = default
return config_value
@property
def seconds_since_update(self):
return (galaxy.model.orm.now.now() - self.update_time).total_seconds()
class Task(object, JobLike):
"""
@@ -1044,6 +1053,33 @@ class ImplicitlyCreatedDatasetCollectionInput(object):
self.input_dataset_collection = input_dataset_collection
class ImplicitCollectionJobs(object):
populated_states = Bunch(
NEW='new', # New implicit jobs object, unpopulated job associations
OK='ok', # Job associations are set and fixed.
FAILED='failed', # There were issues populating job associations, object is in error.
)
def __init__(
self,
id=None,
populated_state=None,
):
self.id = id
self.populated_state = populated_state or ImplicitCollectionJobs.populated_states.NEW
@property
def job_list(self):
return [icjja.job for icjja in self.jobs]
class ImplicitCollectionJobsJobAssociation(object):
def __init__(self):
pass
class PostJobAction(object):
def __init__(self, action_type, workflow_step, output_name=None, action_arguments=None):
self.action_type = action_type
@@ -3208,6 +3244,15 @@ class DatasetCollection(object, Dictifiable, UsesAnnotations):
self.populated_state = DatasetCollection.populated_states.FAILED
self.populated_state_message = message
def finalize(self):
# All jobs have written out their elements - everything should be populated
# but might not be - check that second case! (TODO)
self.mark_as_populated()
if self.has_subcollections:
# THIS IS WRONG - SHOULD ONLY BE TO THE DEPTH OF THE MAP OVER.
for element in self.elements:
element.child_collection.finalize()
@property
def dataset_instances(self):
instances = []
@@ -3308,6 +3353,7 @@ class DatasetCollectionInstance(object, HasName):
populated=self.populated,
populated_state=self.collection.populated_state,
populated_state_message=self.collection.populated_state_message,
element_count=self.collection.element_count,
type="collection", # contents type (distinguished from file or folder (in case of library))
)
@@ -3383,6 +3429,19 @@ class HistoryDatasetCollectionAssociation(DatasetCollectionInstance,
return ((type_coerce(cls.content_type, types.Unicode) + u'-' +
type_coerce(cls.id, types.Unicode)).label('type_id'))
@property
def job_source_type(self):
if self.implicit_collection_jobs_id:
return "ImplicitCollectionJobs"
elif self.job_id:
return "Job"
else:
return None
@property
def job_source_id(self):
return self.implicit_collection_jobs_id or self.job_id
def to_hda_representative(self, multiple=False):
rval = []
for dataset in self.collection.dataset_elements:
@@ -3400,6 +3459,8 @@ class HistoryDatasetCollectionAssociation(DatasetCollectionInstance,
history_content_type=self.history_content_type,
visible=self.visible,
deleted=self.deleted,
job_source_id=self.job_source_id,
job_source_type=self.job_source_type,
**self._base_to_dict(view=view)
)
@@ -3431,6 +3492,11 @@ class HistoryDatasetCollectionAssociation(DatasetCollectionInstance,
name=self.name,
copied_from_history_dataset_collection_association=self,
)
if self.implicit_collection_jobs_id:
hdca.implicit_collection_jobs_id = self.implicit_collection_jobs_id
elif self.job_id:
hdca.job_id = self.job_id
collection_copy = self.collection.copy(
destination=hdca,
element_destination=element_destination,
@@ -3475,6 +3541,8 @@ class DatasetCollectionElement(object, Dictifiable):
dict_collection_visible_keys = ['id', 'element_type', 'element_index', 'element_identifier']
dict_element_visible_keys = ['id', 'element_type', 'element_index', 'element_identifier']
UNINITIALIZED_ELEMENT = object()
def __init__(
self,
id=None,
@@ -3489,7 +3557,7 @@ class DatasetCollectionElement(object, Dictifiable):
self.ldda = element
elif isinstance(element, DatasetCollection):
self.child_collection = element
else:
elif element != self.UNINITIALIZED_ELEMENT:
raise AttributeError('Unknown element type provided: %s' % type(element))
self.id = id
@@ -3507,7 +3575,7 @@ class DatasetCollectionElement(object, Dictifiable):
# TOOD: Rename element_type to element_type.
return "dataset_collection"
else:
raise Exception("Unknown element instance type")
return None
@property
def is_collection(self):
@@ -3522,7 +3590,7 @@ class DatasetCollectionElement(object, Dictifiable):
elif self.child_collection:
return self.child_collection
else:
raise Exception("Unknown element instance type")
return None
@property
def dataset_instance(self):
@@ -3952,7 +4020,7 @@ class StoredWorkflowMenuEntry(object):
self.order_index = None
class WorkflowInvocation(object, Dictifiable):
class WorkflowInvocation(object, UsesCreateAndUpdateTime, Dictifiable):
dict_collection_visible_keys = ['id', 'update_time', 'workflow_id', 'history_id', 'uuid', 'state']
dict_element_visible_keys = ['id', 'update_time', 'workflow_id', 'history_id', 'uuid', 'state']
states = Bunch(
@@ -4027,17 +4095,16 @@ class WorkflowInvocation(object, Dictifiable):
step_invocations = {}
for invocation_step in self.steps:
step_id = invocation_step.workflow_step_id
if step_id not in step_invocations:
step_invocations[step_id] = []
step_invocations[step_id].append(invocation_step)
assert step_id not in step_invocations
step_invocations[step_id] = invocation_step
return step_invocations
def step_invocations_for_step_id(self, step_id):
step_invocations = []
def step_invocation_for_step_id(self, step_id):
target_invocation_step = None
for invocation_step in self.steps:
if step_id == invocation_step.workflow_step_id:
step_invocations.append(invocation_step)
return step_invocations
target_invocation_step = invocation_step
return target_invocation_step
@staticmethod
def poll_active_workflow_ids(
@@ -4063,6 +4130,24 @@ class WorkflowInvocation(object, Dictifiable):
# is relatively intutitive.
return [wid for wid in query.all()]
def add_output(self, workflow_output, step, output_object):
if output_object.history_content_type == "dataset":
output_assoc = WorkflowInvocationOutputDatasetAssociation()
output_assoc.workflow_invocation = self
output_assoc.workflow_output = workflow_output
output_assoc.workflow_step = step
output_assoc.dataset = output_object
self.output_datasets.append(output_assoc)
elif output_object.history_content_type == "dataset_collection":
output_assoc = WorkflowInvocationOutputDatasetCollectionAssociation()
output_assoc.workflow_invocation = self
output_assoc.workflow_output = workflow_output
output_assoc.workflow_step = step
output_assoc.dataset_collection = output_object
self.output_dataset_collections.append(output_assoc)
else:
raise Exception("Uknown output type encountered")
def to_dict(self, view='collection', value_mapper=None, step_details=False):
rval = super(WorkflowInvocation, self).to_dict(view=view, value_mapper=value_mapper)
if view == 'element':
@@ -4078,17 +4163,43 @@ class WorkflowInvocation(object, Dictifiable):
inputs = {}
for step in self.steps:
if step.workflow_step.type == 'tool':
for step_input in step.workflow_step.input_connections:
output_step_type = step_input.output_step.type
if output_step_type in ['data_input', 'data_collection_input']:
src = "hda" if output_step_type == 'data_input' else 'hdca'
for job_input in step.job.input_datasets:
if job_input.name == step_input.input_name:
inputs[str(step_input.output_step.order_index)] = {
"id": job_input.dataset_id, "src": src,
"uuid" : str(job_input.dataset.dataset.uuid) if job_input.dataset.dataset.uuid is not None else None
}
for job in step.jobs:
for step_input in step.workflow_step.input_connections:
output_step_type = step_input.output_step.type
if output_step_type in ['data_input', 'data_collection_input']:
src = "hda" if output_step_type == 'data_input' else 'hdca'
for job_input in job.input_datasets:
if job_input.name == step_input.input_name:
inputs[str(step_input.output_step.order_index)] = {
"id": job_input.dataset_id, "src": src,
"uuid" : str(job_input.dataset.dataset.uuid) if job_input.dataset.dataset.uuid is not None else None
}
rval['inputs'] = inputs
outputs = {}
for output_assoc in self.output_datasets:
label = output_assoc.workflow_output.label
if not label:
continue
outputs[label] = {
'src': 'hda',
'id': output_assoc.dataset_id,
}
output_collections = {}
for output_assoc in self.output_dataset_collections:
label = output_assoc.workflow_output.label
if not label:
continue
output_collections[label] = {
'src': 'hdca',
'id': output_assoc.dataset_collection_id,
}
rval['outputs'] = outputs
rval['output_collections'] = output_collections
return rval
def update(self):
@@ -4121,11 +4232,6 @@ class WorkflowInvocation(object, Dictifiable):
return True
return False
@property
def seconds_since_created(self):
create_time = self.create_time or galaxy.model.orm.now.now() # In case not flushed yet
return (galaxy.model.orm.now.now() - create_time).total_seconds()
class WorkflowInvocationToSubworkflowInvocationAssociation(object, Dictifiable):
dict_collection_visible_keys = ['id', 'workflow_step_id', 'workflow_invocation_id', 'subworkflow_invocation_id']
@@ -4133,36 +4239,83 @@ class WorkflowInvocationToSubworkflowInvocationAssociation(object, Dictifiable):
class WorkflowInvocationStep(object, Dictifiable):
dict_collection_visible_keys = ['id', 'update_time', 'job_id', 'workflow_step_id', 'action']
dict_element_visible_keys = ['id', 'update_time', 'job_id', 'workflow_step_id', 'action']
dict_collection_visible_keys = ['id', 'update_time', 'job_id', 'workflow_step_id', 'state', 'action']
dict_element_visible_keys = ['id', 'update_time', 'job_id', 'workflow_step_id', 'state', 'action']
states = Bunch(
NEW='new', # Brand new workflow invocation step
READY='ready', # Workflow invocation step ready for another iteration of scheduling.
SCHEDULED='scheduled', # Workflow invocation step has been scheduled.
# CANCELLED='cancelled', TODO: implement and expose
# FAILED='failed', TODO: implement and expose
)
def update(self):
self.workflow_invocation.update()
@property
def is_new(self):
return self.state == self.states.NEW
def add_output(self, output_name, output_object):
if output_object.history_content_type == "dataset":
output_assoc = WorkflowInvocationStepOutputDatasetAssociation()
output_assoc.workflow_invocation_step = self
output_assoc.dataset = output_object
output_assoc.output_name = output_name
self.output_datasets.append(output_assoc)
elif output_object.history_content_type == "dataset_collection":
output_assoc = WorkflowInvocationStepOutputDatasetCollectionAssociation()
output_assoc.workflow_invocation_step = self
output_assoc.dataset_collection = output_object
output_assoc.output_name = output_name
self.output_dataset_collections.append(output_assoc)
else:
raise Exception("Uknown output type encountered")
@property
def jobs(self):
if self.job:
return [self.job]
elif self.implicit_collection_jobs:
return self.implicit_collection_jobs.job_list
else:
return []
def to_dict(self, view='collection', value_mapper=None):
rval = super(WorkflowInvocationStep, self).to_dict(view=view, value_mapper=value_mapper)
rval['order_index'] = self.workflow_step.order_index
rval['workflow_step_label'] = self.workflow_step.label
rval['workflow_step_uuid'] = str(self.workflow_step.uuid)
rval['state'] = self.job.state if self.job is not None else None
if self.job is not None and view == 'element':
output_dict = {}
for i in self.job.output_datasets:
if i.dataset is not None:
output_dict[i.name] = {
"id" : i.dataset.id, "src" : "hda",
"uuid" : str(i.dataset.dataset.uuid) if i.dataset.dataset.uuid is not None else None
}
for i in self.job.output_library_datasets:
if i.dataset is not None:
output_dict[i.name] = {
"id" : i.dataset.id, "src" : "ldda",
"uuid" : str(i.dataset.dataset.uuid) if i.dataset.dataset.uuid is not None else None
}
rval['outputs'] = output_dict
# Following no longer makes sense...
# rval['state'] = self.job.state if self.job is not None else None
if view == 'element':
outputs = {}
for output_assoc in self.output_datasets:
name = output_assoc.output_name
outputs[name] = {
'src': 'hda',
'id': output_assoc.dataset.id,
'uuid': str(output_assoc.dataset.dataset.uuid) if output_assoc.dataset.dataset.uuid is not None else None
}
output_collections = {}
for output_assoc in self.output_dataset_collections:
name = output_assoc.output_name
output_collections[name] = {
'src': 'hdca',
'id': output_assoc.dataset_collection.id,
}
rval['outputs'] = outputs
rval['output_collections'] = output_collections
return rval
class WorkflowInvocationStepJobAssociation(object, Dictifiable):
dict_collection_visible_keys = ('id', 'job_id', 'workflow_invocation_step_id')
dict_element_visible_keys = ('id', 'job_id', 'workflow_invocation_step_id')
class WorkflowRequest(object, Dictifiable):
dict_collection_visible_keys = ['id', 'name', 'type', 'state', 'history_id', 'workflow_id']
dict_element_visible_keys = ['id', 'name', 'type', 'state', 'history_id', 'workflow_id']
@@ -4217,6 +4370,26 @@ class WorkflowRequestInputStepParmeter(object, Dictifiable):
dict_collection_visible_keys = ['id', 'workflow_invocation_id', 'workflow_step_id', 'parameter_value']
class WorkflowInvocationOutputDatasetAssociation(object, Dictifiable):
"""Represents links to output datasets for the workflow."""
dict_collection_visible_keys = ['id', 'workflow_invocation_id', 'workflow_step_id', 'dataset_id', 'name']
class WorkflowInvocationOutputDatasetCollectionAssociation(object, Dictifiable):
"""Represents links to output dataset collections for the workflow."""
dict_collection_visible_keys = ['id', 'workflow_invocation_id', 'workflow_step_id', 'dataset_collection_id', 'name']
class WorkflowInvocationStepOutputDatasetAssociation(object, Dictifiable):
"""Represents links to output datasets for the workflow."""
dict_collection_visible_keys = ['id', 'workflow_invocation_step_id', 'dataset_id', 'output_name']
class WorkflowInvocationStepOutputDatasetCollectionAssociation(object, Dictifiable):
"""Represents links to output dataset collections for the workflow."""
dict_collection_visible_keys = ['id', 'workflow_invocation_step_id', 'dataset_collection_id', 'output_name']
class MetadataFile(StorableObject):
def __init__(self, dataset=None, name=None):
+129 -7
View File
@@ -557,6 +557,20 @@ model.ImplicitlyCreatedDatasetCollectionInput.table = Table(
ForeignKey("history_dataset_collection_association.id"), index=True),
Column("name", Unicode(255)))
model.ImplicitCollectionJobs.table = Table(
"implicit_collection_jobs", metadata,
Column("id", Integer, primary_key=True),
Column("populated_state", TrimmedString(64), default='new', nullable=False),
)
model.ImplicitCollectionJobsJobAssociation.table = Table(
"implicit_collection_jobs_job_association", metadata,
Column("id", Integer, primary_key=True),
Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), index=True),
Column("job_id", Integer, ForeignKey("job.id"), index=True), # Consider making this nullable...
Column("order_index", Integer, nullable=False),
)
model.JobExternalOutputMetadata.table = Table(
"job_external_output_metadata", metadata,
Column("id", Integer, primary_key=True),
@@ -702,6 +716,7 @@ model.DatasetCollection.table = Table(
Column("collection_type", Unicode(255), nullable=False),
Column("populated_state", TrimmedString(64), default='ok', nullable=False),
Column("populated_state_message", TEXT),
Column("element_count", Integer, nullable=True),
Column("create_time", DateTime, default=now),
Column("update_time", DateTime, default=now, onupdate=now))
@@ -716,7 +731,10 @@ model.HistoryDatasetCollectionAssociation.table = Table(
Column("deleted", Boolean, default=False),
Column("copied_from_history_dataset_collection_association_id", Integer,
ForeignKey("history_dataset_collection_association.id"), nullable=True),
Column("implicit_output_name", Unicode(255), nullable=True))
Column("implicit_output_name", Unicode(255), nullable=True),
Column("job_id", ForeignKey("job.id"), index=True, nullable=True),
Column("implicit_collection_jobs_id", ForeignKey("implicit_collection_jobs.id"), index=True, nullable=True),
)
model.LibraryDatasetCollectionAssociation.table = Table(
"library_dataset_collection_association", metadata,
@@ -901,9 +919,46 @@ model.WorkflowInvocationStep.table = Table(
Column("update_time", DateTime, default=now, onupdate=now),
Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True, nullable=False),
Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True, nullable=False),
Column("state", TrimmedString(64), index=True),
Column("job_id", Integer, ForeignKey("job.id"), index=True, nullable=True),
Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), index=True, nullable=True),
Column("action", JSONType, nullable=True))
model.WorkflowInvocationOutputDatasetAssociation.table = Table(
"workflow_invocation_output_dataset_association", metadata,
Column("id", Integer, primary_key=True),
Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True),
Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True),
Column("dataset_id", Integer, ForeignKey("history_dataset_association.id"), index=True),
Column("workflow_output_id", Integer, ForeignKey("workflow_output.id"), index=True),
)
model.WorkflowInvocationOutputDatasetCollectionAssociation.table = Table(
"workflow_invocation_output_dataset_collection_association", metadata,
Column("id", Integer, primary_key=True),
Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True),
Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True),
Column("dataset_collection_id", Integer, ForeignKey("history_dataset_collection_association.id"), index=True),
Column("workflow_output_id", Integer, ForeignKey("workflow_output.id"), index=True),
)
model.WorkflowInvocationStepOutputDatasetAssociation.table = Table(
"workflow_invocation_step_output_dataset_association", metadata,
Column("id", Integer, primary_key=True),
Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True),
Column("dataset_id", Integer, ForeignKey("history_dataset_association.id"), index=True),
Column("output_name", String(255), nullable=True),
)
model.WorkflowInvocationStepOutputDatasetCollectionAssociation.table = Table(
"workflow_invocation_step_output_dataset_collection_association", metadata,
Column("id", Integer, primary_key=True),
Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True),
Column("workflow_step_id", Integer, ForeignKey("workflow_step.id"), index=True),
Column("dataset_collection_id", Integer, ForeignKey("history_dataset_collection_association.id"), index=True),
Column("output_name", String(255), nullable=True),
)
model.WorkflowInvocationToSubworkflowInvocationAssociation.table = Table(
"workflow_invocation_to_subworkflow_invocation_association", metadata,
Column("id", Integer, primary_key=True),
@@ -2066,6 +2121,31 @@ simple_mapping(model.ImplicitlyCreatedDatasetCollectionInput,
),
)
simple_mapping(model.ImplicitCollectionJobs)
# simple_mapping(
# model.ImplicitCollectionJobsHistoryDatasetCollectionAssociation,
# history_dataset_collection_associations=relation(
# model.HistoryDatasetCollectionAssociation,
# backref=backref("implicit_collection_jobs_association", uselist=False),
# uselist=True,
# ),
# )
simple_mapping(
model.ImplicitCollectionJobsJobAssociation,
implicit_collection_jobs=relation(
model.ImplicitCollectionJobs,
backref=backref("jobs", uselist=True),
uselist=False,
),
job=relation(
model.Job,
backref=backref("implicit_collection_jobs_association", uselist=False),
uselist=False,
),
)
mapper(model.JobParameter, model.JobParameter.table)
mapper(model.JobExternalOutputMetadata, model.JobExternalOutputMetadata.table, properties=dict(
@@ -2157,6 +2237,16 @@ simple_mapping(model.HistoryDatasetCollectionAssociation,
model.ImplicitlyCreatedDatasetCollectionInput.table.c.dataset_collection_id)),
backref="dataset_collection",
),
implicit_collection_jobs=relation(
model.ImplicitCollectionJobs,
backref=backref("history_dataset_collection_associations", uselist=True),
uselist=False,
),
job=relation(
model.Job,
backref=backref("history_dataset_collection_associations", uselist=True),
uselist=False,
),
tags=relation(model.HistoryDatasetCollectionTagAssociation,
order_by=model.HistoryDatasetCollectionTagAssociation.table.c.id,
backref='dataset_collections'),
@@ -2310,7 +2400,7 @@ mapper(model.WorkflowInvocation, model.WorkflowInvocation.table, properties=dict
uselist=True,
),
steps=relation(model.WorkflowInvocationStep,
backref='workflow_invocation'),
backref="workflow_invocation"),
workflow=relation(model.Workflow)
))
@@ -2323,12 +2413,11 @@ mapper(model.WorkflowInvocationToSubworkflowInvocationAssociation, model.Workflo
workflow_step=relation(model.WorkflowStep),
))
mapper(model.WorkflowInvocationStep, model.WorkflowInvocationStep.table, properties=dict(
simple_mapping(model.WorkflowInvocationStep,
workflow_step=relation(model.WorkflowStep),
job=relation(model.Job,
backref=backref('workflow_invocation_step',
uselist=False))
))
job=relation(model.Job, backref=backref('workflow_invocation_step', uselist=False), uselist=False),
implicit_collection_jobs=relation(model.ImplicitCollectionJobs, backref=backref('workflow_invocation_step', uselist=False), uselist=False),)
simple_mapping(model.WorkflowRequestInputParameter,
workflow_invocation=relation(model.WorkflowInvocation))
@@ -2358,6 +2447,39 @@ mapper(model.MetadataFile, model.MetadataFile.table, properties=dict(
library_dataset=relation(model.LibraryDatasetDatasetAssociation)
))
simple_mapping(
model.WorkflowInvocationOutputDatasetAssociation,
workflow_invocation=relation(model.WorkflowInvocation, backref="output_datasets"),
workflow_step=relation(model.WorkflowStep),
dataset=relation(model.HistoryDatasetAssociation),
workflow_output=relation(model.WorkflowOutput),
)
simple_mapping(
model.WorkflowInvocationOutputDatasetCollectionAssociation,
workflow_invocation=relation(model.WorkflowInvocation, backref="output_dataset_collections"),
workflow_step=relation(model.WorkflowStep),
dataset_collection=relation(model.HistoryDatasetCollectionAssociation),
workflow_output=relation(model.WorkflowOutput),
)
simple_mapping(
model.WorkflowInvocationStepOutputDatasetAssociation,
workflow_invocation_step=relation(model.WorkflowInvocationStep, backref="output_datasets"),
dataset=relation(model.HistoryDatasetAssociation),
)
simple_mapping(
model.WorkflowInvocationStepOutputDatasetCollectionAssociation,
workflow_invocation_step=relation(model.WorkflowInvocationStep, backref="output_dataset_collections"),
dataset_collection=relation(model.HistoryDatasetCollectionAssociation),
)
mapper(model.PageRevision, model.PageRevision.table)
mapper(model.Page, model.Page.table, properties=dict(
@@ -0,0 +1,172 @@
"""
Migration script for collections and workflows connections.
"""
from __future__ import print_function
import datetime
import logging
from collections import OrderedDict
from sqlalchemy import Column, ForeignKey, Integer, MetaData, String, Table
from galaxy.model.custom_types import TrimmedString
now = datetime.datetime.utcnow
log = logging.getLogger(__name__)
metadata = MetaData()
workflow_invocation_output_dataset_association_table = Table(
"workflow_invocation_output_dataset_association", metadata,
Column("id", Integer, primary_key=True),
Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True),
Column("workflow_step_id", Integer, ForeignKey("workflow_step.id")),
Column("dataset_id", Integer, ForeignKey("history_dataset_association.id"), index=True),
Column("workflow_output_id", Integer, ForeignKey("workflow_output.id")),
)
workflow_invocation_output_dataset_collection_association_table = Table(
"workflow_invocation_output_dataset_collection_association", metadata,
Column("id", Integer, primary_key=True),
Column("workflow_invocation_id", Integer, ForeignKey("workflow_invocation.id"), index=True),
Column("workflow_step_id", Integer, ForeignKey("workflow_step.id")),
Column("dataset_collection_id", Integer, ForeignKey("history_dataset_collection_association.id"), index=True),
Column("workflow_output_id", Integer, ForeignKey("workflow_output.id")),
)
workflow_invocation_step_output_dataset_association_table = Table(
"workflow_invocation_step_output_dataset_association", metadata,
Column("id", Integer, primary_key=True),
Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True),
Column("dataset_id", Integer, ForeignKey("history_dataset_association.id"), index=True),
Column("output_name", String(255), nullable=True),
)
workflow_invocation_step_output_dataset_collection_association_table = Table(
"workflow_invocation_step_output_dataset_collection_association", metadata,
Column("id", Integer, primary_key=True),
Column("workflow_invocation_step_id", Integer, ForeignKey("workflow_invocation_step.id"), index=True),
Column("workflow_step_id", Integer, ForeignKey("workflow_step.id")),
Column("dataset_collection_id", Integer, ForeignKey("history_dataset_collection_association.id"), index=True),
Column("output_name", String(255), nullable=True),
)
implicit_collection_jobs_table = Table(
"implicit_collection_jobs", metadata,
Column("id", Integer, primary_key=True),
Column("populated_state", TrimmedString(64), default='new', nullable=False),
)
implicit_collection_jobs_job_association_table = Table(
"implicit_collection_jobs_job_association", metadata,
Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), index=True),
Column("id", Integer, primary_key=True),
Column("job_id", Integer, ForeignKey("job.id"), index=True), # Consider making this nullable...
Column("order_index", Integer, nullable=False),
)
def get_new_tables():
# Normally we define this globally in the file, but we need to delay the
# reading of existing tables because an existing workflow_invocation_step
# table exists that we want to recreate.
tables = OrderedDict()
tables["workflow_invocation_output_dataset_association"] = workflow_invocation_output_dataset_association_table
tables["workflow_invocation_output_dataset_collection_association"] = workflow_invocation_output_dataset_collection_association_table
tables["workflow_invocation_step_output_dataset_association"] = workflow_invocation_step_output_dataset_association_table
tables["workflow_invocation_step_output_dataset_collection_association"] = workflow_invocation_step_output_dataset_collection_association_table
tables["implicit_collection_jobs"] = implicit_collection_jobs_table
tables["implicit_collection_jobs_job_association"] = implicit_collection_jobs_job_association_table
return tables
def upgrade(migrate_engine):
metadata.bind = migrate_engine
print(__doc__)
metadata.reflect()
tables = get_new_tables()
for table in tables.values():
__create(table)
def nextval(table, col='id'):
if migrate_engine.name in ['postgres', 'postgresql']:
return "nextval('%s_%s_seq')" % (table, col)
elif migrate_engine.name in ['mysql', 'sqlite']:
return "null"
else:
raise Exception("Unhandled database type")
# Set default for creation to scheduled, actual mapping has new as default.
workflow_invocation_step_state_column = Column("state", TrimmedString(64), default="scheduled")
if migrate_engine.name in ['postgres', 'postgresql']:
implicit_collection_jobs_id_column = Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), nullable=True)
job_id_column = Column("job_id", Integer, ForeignKey("job.id"), nullable=True)
else:
implicit_collection_jobs_id_column = Column("implicit_collection_jobs_id", Integer, nullable=True)
job_id_column = Column("job_id", Integer, nullable=True)
dataset_collection_element_count_column = Column("element_count", Integer, nullable=True)
__add_column(implicit_collection_jobs_id_column, "history_dataset_collection_association", metadata)
__add_column(job_id_column, "history_dataset_collection_association", metadata)
__add_column(dataset_collection_element_count_column, "dataset_collection", metadata)
implicit_collection_jobs_id_column = Column("implicit_collection_jobs_id", Integer, ForeignKey("implicit_collection_jobs.id"), nullable=True)
__add_column(implicit_collection_jobs_id_column, "workflow_invocation_step", metadata)
__add_column(workflow_invocation_step_state_column, "workflow_invocation_step", metadata)
cmd = \
"UPDATE dataset_collection SET element_count = " + \
"(SELECT (CASE WHEN count(*) > 0 THEN count(*) ELSE 0 END) FROM dataset_collection_element WHERE " + \
"dataset_collection_element.dataset_collection_id = dataset_collection.id)"
migrate_engine.execute(cmd)
def __add_column(column, table_name, metadata, **kwds):
try:
table = Table(table_name, metadata, autoload=True)
column.create(table, **kwds)
except Exception:
log.exception("Adding column %s failed.", column)
def __drop_column(column_name, table_name, metadata):
try:
table = Table(table_name, metadata, autoload=True)
getattr(table.c, column_name).drop()
except Exception:
log.exception("Dropping column %s failed.", column_name)
def downgrade(migrate_engine):
metadata.bind = migrate_engine
metadata.reflect()
__drop_column("implicit_collection_jobs_id", "history_dataset_collection_association", metadata)
__drop_column("job_id", "history_dataset_collection_association", metadata)
__drop_column("implicit_collection_jobs_id", "workflow_invocation_step", metadata)
__drop_column("state", "workflow_invocation_step", metadata)
__drop_column("element_count", "dataset_collection", metadata)
tables = get_new_tables()
for table in reversed(tables.values()):
__drop(table)
def __create(table):
try:
table.create()
except Exception:
log.exception("Creating %s table failed.", table.name)
def __drop(table):
try:
table.drop()
except Exception:
log.exception("Dropping %s table failed.", table.name)
+31 -16
View File
@@ -84,7 +84,10 @@ from galaxy.web import url_for
from galaxy.web.form_builder import SelectField
from galaxy.work.context import WorkRequestContext
from tool_shed.util import common_util
from .execute import execute as execute_job
from .execute import (
execute as execute_job,
MappingParameters,
)
from .loader import (
imported_macro_paths,
raw_tool_xml_tree,
@@ -1244,8 +1247,6 @@ class Tool(object, Dictifiable):
# Fixed set of input parameters may correspond to any number of jobs.
# Expand these out to individual parameters for given jobs (tool executions).
expanded_incomings, collection_info = expand_meta_parameters(trans, self, incoming)
if not expanded_incomings:
raise exceptions.MessageException('Tool execution failed, trying to run a tool over an empty collection.')
# Remapping a single job to many jobs doesn't make sense, so disable
# remap if multi-runs of tools are being used.
@@ -1295,24 +1296,38 @@ class Tool(object, Dictifiable):
err_data = {key: value for d in all_errors for (key, value) in d.items()}
raise exceptions.MessageException(', '.join(msg for msg in err_data.values()), err_data=err_data)
else:
execution_tracker = execute_job(trans, self, all_params, history=request_context.history, rerun_remap_job_id=rerun_remap_job_id, collection_info=collection_info)
if execution_tracker.successful_jobs:
return dict(out_data=execution_tracker.output_datasets,
num_jobs=len(execution_tracker.successful_jobs),
job_errors=execution_tracker.execution_errors,
jobs=execution_tracker.successful_jobs,
output_collections=execution_tracker.output_collections,
implicit_collections=execution_tracker.implicit_collections)
else:
mapping_params = MappingParameters(incoming, all_params)
execution_tracker = execute_job(trans, self, mapping_params, history=request_context.history, rerun_remap_job_id=rerun_remap_job_id, collection_info=collection_info)
# Raise an exception if there were jobs to execute and none of them were submitted,
# if at least one is submitted or there are no jobs to execute - return aggregate
# information including per-job errors. Arguably we should just always return the
# aggregate information - we just haven't done that historically.
raise_execution_exception = not execution_tracker.successful_jobs and len(all_params) > 0
if raise_execution_exception:
raise exceptions.MessageException(execution_tracker.execution_errors[0])
def handle_single_execution(self, trans, rerun_remap_job_id, params, history, mapping_over_collection, execution_cache=None):
return dict(out_data=execution_tracker.output_datasets,
num_jobs=len(execution_tracker.successful_jobs),
job_errors=execution_tracker.execution_errors,
jobs=execution_tracker.successful_jobs,
output_collections=execution_tracker.output_collections,
implicit_collections=execution_tracker.implicit_collections)
def handle_single_execution(self, trans, rerun_remap_job_id, execution_slice, history, execution_cache=None):
"""
Return a pair with whether execution is successful as well as either
resulting output data or an error message indicating the problem.
"""
try:
job, out_data = self.execute(trans, incoming=params, history=history, rerun_remap_job_id=rerun_remap_job_id, mapping_over_collection=mapping_over_collection, execution_cache=execution_cache)
job, out_data = self.execute(
trans,
incoming=execution_slice.param_combination,
history=history,
rerun_remap_job_id=rerun_remap_job_id,
execution_cache=execution_cache,
dataset_collection_elements=execution_slice.dataset_collection_elements,
)
except httpexceptions.HTTPFound as e:
# if it's a paste redirect exception, pass it up the stack
raise e
@@ -2255,7 +2270,7 @@ class DatabaseOperationTool(Tool):
def check_inputs_ready(self, input_datasets, input_dataset_collections):
def check_dataset_instance(input_dataset):
if input_dataset.is_pending:
raise ToolInputsNotReadyException()
raise ToolInputsNotReadyException("An input dataset is pending.")
if self.require_dataset_ok:
if input_dataset.state != input_dataset.dataset.states.OK:
@@ -2267,7 +2282,7 @@ class DatabaseOperationTool(Tool):
for input_dataset_collection_pairs in input_dataset_collections.values():
for input_dataset_collection, is_mapped in input_dataset_collection_pairs:
if not input_dataset_collection.collection.populated:
raise ToolInputsNotReadyException()
raise ToolInputsNotReadyException("An input collection is not populated.")
map(check_dataset_instance, input_dataset_collection.dataset_instances)
+15 -7
View File
@@ -195,7 +195,7 @@ class DefaultToolAction(object):
return history, inp_data, inp_dataset_collections
def execute(self, tool, trans, incoming={}, return_job=False, set_output_hid=True, history=None, job_params=None, rerun_remap_job_id=None, mapping_over_collection=False, execution_cache=None):
def execute(self, tool, trans, incoming={}, return_job=False, set_output_hid=True, history=None, job_params=None, rerun_remap_job_id=None, execution_cache=None, dataset_collection_elements=None):
"""
Executes a tool, creating job and tool outputs, associating them, and
submitting the job to the job queue. If history is not specified, use
@@ -267,7 +267,7 @@ class DefaultToolAction(object):
tool=tool,
tool_action=self,
input_collections=input_collections,
mapping_over_collection=mapping_over_collection,
dataset_collection_elements=dataset_collection_elements,
on_text=on_text,
incoming=incoming,
params=wrapped_params.params,
@@ -304,8 +304,12 @@ class DefaultToolAction(object):
data = app.model.HistoryDatasetAssociation(extension=ext, create_dataset=True, flush=False)
if hidden is None:
hidden = output.hidden
if not hidden and dataset_collection_elements is not None: # Mapping over a collection - hide datasets
hidden = True
if hidden:
data.visible = False
if dataset_collection_elements is not None and name in dataset_collection_elements:
dataset_collection_elements[name].hda = data
trans.sa_session.add(data)
trans.app.security_agent.set_all_dataset_permissions(data.dataset, output_permissions, new=True)
for _, tag in preserved_tags.items():
@@ -351,6 +355,7 @@ class DefaultToolAction(object):
for name, output in tool.outputs.items():
if not filter_output(output, incoming):
handle_output_timer = ExecutionTimer()
if output.collection:
collections_manager = app.dataset_collections_service
element_identifiers = []
@@ -402,15 +407,14 @@ class DefaultToolAction(object):
element_kwds = dict(elements=collections_manager.ELEMENTS_UNINITIALIZED)
else:
element_kwds = dict(element_identifiers=element_identifiers)
output_collections.create_collection(
output=output,
name=name,
tags=preserved_tags,
**element_kwds
)
log.info("Handled collection output named %s for tool %s %s" % (name, tool.id, handle_output_timer))
else:
handle_output_timer = ExecutionTimer()
handle_output(name, output)
log.info("Handled output named %s for tool %s %s" % (name, tool.id, handle_output_timer))
@@ -594,6 +598,7 @@ class DefaultToolAction(object):
job.add_implicit_output_dataset_collection(name, dataset_collection)
for name, dataset_collection_instance in out_collection_instances.items():
job.add_output_dataset_collection(name, dataset_collection_instance)
dataset_collection_instance.job = job
def _check_input_data_access(self, trans, job, inp_data, current_user_roles):
access_timer = ExecutionTimer()
@@ -670,13 +675,13 @@ class OutputCollections(object):
parameter).
"""
def __init__(self, trans, history, tool, tool_action, input_collections, mapping_over_collection, on_text, incoming, params, job_params):
def __init__(self, trans, history, tool, tool_action, input_collections, dataset_collection_elements, on_text, incoming, params, job_params):
self.trans = trans
self.history = history
self.tool = tool
self.tool_action = tool_action
self.input_collections = input_collections
self.mapping_over_collection = mapping_over_collection
self.dataset_collection_elements = dataset_collection_elements
self.on_text = on_text
self.incoming = incoming
self.params = params
@@ -710,12 +715,15 @@ class OutputCollections(object):
for dataset in value.dataset_instances:
assert dataset.history is not None
if self.mapping_over_collection:
if self.dataset_collection_elements is not None:
dc = collections_manager.create_dataset_collection(
self.trans,
collection_type=collection_type,
**element_kwds
)
if name in self.dataset_collection_elements:
self.dataset_collection_elements[name].child_collection = dc
# self.trans.sa_session.add(self.dataset_collection_elements[name])
self.out_collections[name] = dc
else:
hdca_name = self.tool_action.get_output_name(
+2 -2
View File
@@ -21,7 +21,7 @@ class ModelOperationToolAction(DefaultToolAction):
tool.check_inputs_ready(inp_data, inp_dataset_collections)
def execute(self, tool, trans, incoming={}, set_output_hid=False, overwrite=True, history=None, job_params=None, mapping_over_collection=False, execution_cache=None, **kwargs):
def execute(self, tool, trans, incoming={}, set_output_hid=False, overwrite=True, history=None, job_params=None, execution_cache=None, **kwargs):
if execution_cache is None:
execution_cache = ToolExecutionCache(trans)
@@ -42,7 +42,7 @@ class ModelOperationToolAction(DefaultToolAction):
tool=tool,
tool_action=self,
input_collections=input_collections,
mapping_over_collection=mapping_over_collection,
dataset_collection_elements=kwargs.get("dataset_collection_elements", None),
on_text=on_text,
incoming=incoming,
params=wrapped_params.params,
+336 -116
View File
@@ -4,12 +4,17 @@ from various states, tracking results, and building implicit dataset
collections from matched collections.
"""
import collections
import itertools
import logging
from threading import Thread
import six
from six.moves.queue import Queue
from galaxy.tools.actions import on_text_for_names, ToolExecutionCache
from galaxy import model
from galaxy.dataset_collections.structure import tool_output_to_structure
from galaxy.tools.actions import filter_output, on_text_for_names, ToolExecutionCache
from galaxy.tools.parser import ToolOutputCollectionPart
from galaxy.util import ExecutionTimer
@@ -18,36 +23,51 @@ log = logging.getLogger(__name__)
EXECUTION_SUCCESS_MESSAGE = "Tool [%s] created job [%s] %s"
def execute(trans, tool, param_combinations, history, rerun_remap_job_id=None, collection_info=None, workflow_invocation_uuid=None):
class PartialJobExecution(Exception):
def __init__(self, execution_tracker):
self.execution_tracker = execution_tracker
MappingParameters = collections.namedtuple("MappingParameters", ["param_template", "param_combinations"])
def execute(trans, tool, mapping_params, history, rerun_remap_job_id=None, collection_info=None, workflow_invocation_uuid=None, invocation_step=None, max_num_jobs=None, job_callback=None):
"""
Execute a tool and return object containing summary (output data, number of
failures, etc...).
"""
if max_num_jobs:
assert invocation_step is not None
if rerun_remap_job_id:
assert invocation_step is None
all_jobs_timer = ExecutionTimer()
execution_tracker = ToolExecutionTracker(tool, param_combinations, collection_info)
if invocation_step is None:
execution_tracker = ToolExecutionTracker(trans, tool, mapping_params, collection_info)
else:
execution_tracker = WorkflowStepExecutionTracker(trans, tool, mapping_params, collection_info, invocation_step, job_callback=job_callback)
app = trans.app
execution_cache = ToolExecutionCache(trans)
def execute_single_job(params):
def execute_single_job(execution_slice):
job_timer = ExecutionTimer()
params = execution_slice.param_combination
if workflow_invocation_uuid:
params['__workflow_invocation_uuid__'] = workflow_invocation_uuid
elif '__workflow_invocation_uuid__' in params:
# Only workflow invocation code gets to set this, ignore user supplied
# values or rerun parameters.
del params['__workflow_invocation_uuid__']
job, result = tool.handle_single_execution(trans, rerun_remap_job_id, params, history, collection_info, execution_cache)
job, result = tool.handle_single_execution(trans, rerun_remap_job_id, execution_slice, history, execution_cache)
if job:
message = EXECUTION_SUCCESS_MESSAGE % (tool.id, job.id, job_timer)
log.debug(message)
execution_tracker.record_success(job, result)
execution_tracker.record_success(execution_slice, job, result)
else:
execution_tracker.record_error(result)
config = app.config
burst_at = getattr(config, 'tool_submission_burst_at', 10)
burst_threads = getattr(config, 'tool_submission_burst_threads', 1)
tool_action = tool.tool_action
if hasattr(tool_action, "check_inputs_ready"):
for params in execution_tracker.param_combinations:
@@ -59,11 +79,26 @@ def execute(trans, tool, param_combinations, history, rerun_remap_job_id=None, c
history
)
execution_tracker.ensure_implicit_collections_populated(history, mapping_params.param_template)
config = app.config
burst_at = getattr(config, 'tool_submission_burst_at', 10)
burst_threads = getattr(config, 'tool_submission_burst_threads', 1)
job_count = len(execution_tracker.param_combinations)
if job_count < burst_at or burst_threads < 2:
for params in execution_tracker.param_combinations:
execute_single_job(params)
jobs_executed = 0
has_remaining_jobs = False
if (job_count < burst_at or burst_threads < 2):
for execution_slice in execution_tracker.new_execution_slices():
if max_num_jobs and jobs_executed >= max_num_jobs:
has_remaining_jobs = True
break
else:
execute_single_job(execution_slice)
jobs_executed += 1
else:
# TODO: re-record success...
q = Queue()
def worker():
@@ -77,42 +112,270 @@ def execute(trans, tool, param_combinations, history, rerun_remap_job_id=None, c
t.daemon = True
t.start()
for params in execution_tracker.param_combinations:
q.put(params)
for execution_slice in execution_tracker.new_execution_slices():
if max_num_jobs and jobs_executed >= max_num_jobs:
has_remaining_jobs = True
break
else:
q.put(execution_slice)
jobs_executed += 1
q.join()
log.debug("Executed %d job(s) for tool %s request: %s" % (job_count, tool.id, all_jobs_timer))
if collection_info:
history = history or tool.get_default_history_by_trans(trans)
if len(param_combinations) == 0:
template = "Attempting to map over an empty collection, this is not yet implemented. collection_info is [%s]"
message = template % collection_info
log.warning(message)
raise Exception(message)
params = param_combinations[0]
execution_tracker.create_output_collections(trans, history, params)
if has_remaining_jobs:
raise PartialJobExecution(execution_tracker)
else:
execution_tracker.finalize_dataset_collections(trans)
log.debug("Executed %d job(s) for tool %s request: %s" % (job_count, tool.id, all_jobs_timer))
return execution_tracker
class ToolExecutionTracker(object):
class ExecutionSlice(object):
def __init__(self, tool, param_combinations, collection_info):
def __init__(self, job_index, param_combination, dataset_collection_elements=None):
self.job_index = job_index
self.param_combination = param_combination
self.dataset_collection_elements = dataset_collection_elements
class ExecutionTracker(object):
def __init__(self, trans, tool, mapping_params, collection_info):
# Known ahead of time...
self.trans = trans
self.tool = tool
self.param_combinations = param_combinations
self.mapping_params = mapping_params
self.collection_info = collection_info
self.successful_jobs = []
self._on_text = None
# Populated as we go...
self.failed_jobs = 0
self.execution_errors = []
self.successful_jobs = []
self.output_datasets = []
self.output_collections = []
self.outputs_by_output_name = collections.defaultdict(list)
self.implicit_collections = {}
def record_success(self, job, outputs):
@property
def param_combinations(self):
return self.mapping_params.param_combinations
@property
def example_params(self):
if self.mapping_params.param_combinations:
return self.mapping_params.param_combinations[0]
else:
# TODO: This isn't quite right - what we want is something like param_template wrapped,
# need a test case with an output filter applied to an empty list, still this is
# an improvement over not allowing mapping of empty lists.
return self.mapping_params.param_template
@property
def job_count(self):
return len(self.param_combinations)
def record_error(self, error):
self.failed_jobs += 1
message = "There was a failure executing a job for tool [%s] - %s"
log.warning(message, self.tool.id, error)
self.execution_errors.append(error)
@property
def on_text(self):
if self._on_text is None:
collection_names = ["collection %d" % c.hid for c in self.collection_info.collections.values()]
self._on_text = on_text_for_names(collection_names)
return self._on_text
def output_name(self, trans, history, params, output):
on_text = self.on_text
try:
output_collection_name = self.tool.tool_action.get_output_name(
output,
dataset=None,
tool=self.tool,
on_text=on_text,
trans=trans,
history=history,
params=params,
incoming=None,
job_params=None,
)
except Exception:
output_collection_name = "%s across %s" % (self.tool.name, on_text)
return output_collection_name
def sliced_input_collection_type(self, input_name):
if self.is_implicit_input(input_name):
subcollection_mapping_type = self.collection_info.subcollection_mapping_type(input_name)
return subcollection_mapping_type
# return self.collection_info.structure.sliced_input_collection_type(self.implicit_inputs[input_name])
else:
return self.mapping_params.param_template[input_name].collection.collection_type
def _structure_for_output(self, trans, tool_output):
structure = self.collection_info.structure
if hasattr(tool_output, "default_identifier_source"):
# Switch the structure for outputs if the output specified a default_identifier_source
collection_type_descriptions = trans.app.dataset_collections_service.collection_type_descriptions
source_collection = self.collection_info.collections.get(tool_output.default_identifier_source)
if source_collection:
collection_type_description = collection_type_descriptions.for_collection_type(source_collection.collection.collection_type)
_structure = structure.for_dataset_collection(source_collection.collection, collection_type_description=collection_type_description)
if structure.can_match(_structure):
structure = _structure
return structure
def _element_identifiers_for_output(self, trans, tool_output, outputs):
output_structure = self._structure_for_output(trans, tool_output)
element_identifiers = output_structure.element_identifiers_for_outputs(trans, outputs)
return element_identifiers
def _mapped_output_structure(self, trans, tool_output):
collections_manager = trans.app.dataset_collections_service
output_structure = tool_output_to_structure(self.sliced_input_collection_type, tool_output, collections_manager)
mapping_structure = self._structure_for_output(trans, tool_output)
# Output structure may not be known, but input structure must be,
# otherwise this step of the workflow shouldn't have been scheduled
# or the tool should not have been executable on this input.
mapped_output_structure = mapping_structure.multiply(output_structure, uninitialized=True)
return mapped_output_structure
def ensure_implicit_collections_populated(self, history, params):
if not self.collection_info:
return
history = history or self.tool.get_default_history_by_trans(self.trans)
# params = param_combinations[0] if param_combinations else mapping_params.param_template
self.precreate_output_collections(history, params)
def precreate_output_collections(self, history, params):
# params is just one sample tool param execution with parallelized
# collection replaced with a specific dataset. Need to replace this
# with the collection and wrap everything up so can evaluate output
# label.
trans = self.trans
params.update(self.collection_info.collections) # Replace datasets with source collections for labelling outputs.
collection_instances = {}
implicit_inputs = self.implicit_inputs
implicit_collection_jobs = model.ImplicitCollectionJobs()
for output_name, output in self.tool.outputs.items():
if filter_output(output, self.example_params):
continue
output_collection_name = self.output_name(trans, history, params, output)
effective_structure = self._mapped_output_structure(trans, output)
collection_instance = trans.app.dataset_collections_service.precreate_dataset_collection_instance(
trans=trans,
parent=history,
name=output_collection_name,
implicit_inputs=implicit_inputs,
implicit_output_name=output_name,
structure=effective_structure,
)
collection_instance.implicit_collection_jobs = implicit_collection_jobs
collection_instances[output_name] = collection_instance
trans.sa_session.add(collection_instance)
# Needed to flush the association created just above with
# job.add_output_dataset_collection.
trans.sa_session.flush()
self.implicit_collections = collection_instances
@property
def implicit_collection_jobs(self):
# TODO: refactor to track this properly maybe?
if self.implicit_collections:
return six.next(six.itervalues(self.implicit_collections)).implicit_collection_jobs
else:
return None
def finalize_dataset_collections(self, trans):
# TODO: this probably needs to be reworked some, we should have the collection methods
# return a list of changed objects to add to the session and flush and we should only
# be finalizing collections to a depth of self.collection_info.structure. So for instance
# if you are mapping a list over a tool that dynamically generates lists - we won't actually
# know the structure of the inner list until after its job is complete.
if self.failed_jobs > 0:
for i, implicit_collection in enumerate(self.implicit_collections.values()):
if i == 0:
implicit_collection_jobs = implicit_collection.implicit_collection_jobs
implicit_collection_jobs.populated_state = "failed"
trans.sa_session.add(implicit_collection_jobs)
implicit_collection.collection.handle_population_failed("One or more jobs failed during dataset initialization.")
trans.sa_session.add(implicit_collection.collection)
else:
for i, implicit_collection in enumerate(self.implicit_collections.values()):
if i == 0:
implicit_collection_jobs = implicit_collection.implicit_collection_jobs
implicit_collection_jobs.populated_state = "ok"
trans.sa_session.add(implicit_collection_jobs)
implicit_collection.collection.finalize()
trans.sa_session.add(implicit_collection.collection)
trans.sa_session.flush()
@property
def implicit_inputs(self):
implicit_inputs = list(self.collection_info.collections.items())
return implicit_inputs
def is_implicit_input(self, input_name):
return input_name in self.collection_info.collections
def walk_implicit_collections(self):
return self.collection_info.structure.walk_collections(self.implicit_collections)
def new_execution_slices(self):
if self.collection_info is None:
for job_index, param_combination in enumerate(self.param_combinations):
yield ExecutionSlice(job_index, param_combination)
else:
for execution_slice in self.new_collection_execution_slices():
yield execution_slice
def record_success(self, execution_slice, job, outputs):
# TODO: successful_jobs need to be inserted in the correct place...
self.successful_jobs.append(job)
self.output_datasets.extend(outputs)
for job_output in job.output_dataset_collection_instances:
self.output_collections.append((job_output.name, job_output.dataset_collection_instance))
if self.implicit_collections:
implicit_collection_jobs = None
for output_name, collection_instance in self.implicit_collections.items():
job.add_output_dataset_collection(output_name, collection_instance)
if implicit_collection_jobs is None:
implicit_collection_jobs = collection_instance.implicit_collection_jobs
job_assoc = model.ImplicitCollectionJobsJobAssociation()
job_assoc.order_index = execution_slice.job_index
job_assoc.implicit_collection_jobs = implicit_collection_jobs
job_assoc.job_id = job.id
self.trans.sa_session.add(job_assoc)
# Seperate these because workflows need to track their jobs belong to the invocation
# in the database immediately and they can be recovered.
class ToolExecutionTracker(ExecutionTracker):
def __init__(self, trans, tool, mapping_params, collection_info):
super(ToolExecutionTracker, self).__init__(trans, tool, mapping_params, collection_info)
# New to track these things for tool output API response in the tool case,
# in the workflow case we just write stuff to the database and forget about
# it.
self.outputs_by_output_name = collections.defaultdict(list)
def record_success(self, execution_slice, job, outputs):
super(ToolExecutionTracker, self).record_success(execution_slice, job, outputs)
for output_name, output_dataset in outputs:
if ToolOutputCollectionPart.is_named_collection_part_name(output_name):
# Skip known collection outputs, these will be covered by
@@ -121,101 +384,58 @@ class ToolExecutionTracker(object):
self.outputs_by_output_name[output_name].append(output_dataset)
for job_output in job.output_dataset_collections:
self.outputs_by_output_name[job_output.name].append(job_output.dataset_collection)
for job_output in job.output_dataset_collection_instances:
self.output_collections.append((job_output.name, job_output.dataset_collection_instance))
def record_error(self, error):
self.failed_jobs += 1
message = "There was a failure executing a job for tool [%s] - %s"
log.warning(message, self.tool.id, error)
self.execution_errors.append(error)
def new_collection_execution_slices(self):
for job_index, (param_combination, dataset_collection_elements) in enumerate(itertools.izip(self.param_combinations, self.walk_implicit_collections())):
for dataset_collection_element in dataset_collection_elements.values():
assert dataset_collection_element.element_object is None
def create_output_collections(self, trans, history, params):
# TODO: Move this function - it doesn't belong here but it does need
# the information in this class and potential extensions.
if self.failed_jobs > 0:
return []
yield ExecutionSlice(job_index, param_combination, dataset_collection_elements)
structure = self.collection_info.structure
# params is just one sample tool param execution with parallelized
# collection replaced with a specific dataset. Need to replace this
# with the collection and wrap everything up so can evaluate output
# label.
params.update(self.collection_info.collections) # Replace datasets with source collections for labelling outputs.
class WorkflowStepExecutionTracker(ExecutionTracker):
collection_names = ["collection %d" % c.hid for c in self.collection_info.collections.values()]
on_text = on_text_for_names(collection_names)
def __init__(self, trans, tool, mapping_params, collection_info, invocation_step, job_callback):
super(WorkflowStepExecutionTracker, self).__init__(trans, tool, mapping_params, collection_info)
self.invocation_step = invocation_step
self.job_callback = job_callback
collections = {}
def record_success(self, execution_slice, job, outputs):
super(WorkflowStepExecutionTracker, self).record_success(execution_slice, job, outputs)
if self.collection_info:
self.invocation_step.implicit_collection_jobs = self.implicit_collection_jobs
else:
self.invocation_step.job = job
self.job_callback(job)
implicit_inputs = list(self.collection_info.collections.items())
for output_name, outputs in self.outputs_by_output_name.items():
if not len(structure) == len(outputs):
# Output does not have the same structure, if all jobs were
# successfully submitted this shouldn't have happened.
log.warning("Problem matching up datasets while attempting to create implicit dataset collections")
def new_collection_execution_slices(self):
for job_index, (param_combination, dataset_collection_elements) in enumerate(itertools.izip(self.param_combinations, self.walk_implicit_collections())):
# Two options here - check if the element has been populated or check if the
# a WorkflowInvocationStepJobAssociation exists. Not sure which is better but
# for now I have the first so lets check.
found_result = False
for dataset_collection_element in dataset_collection_elements.values():
if dataset_collection_element.element_object is not None:
found_result = True
break
if found_result:
continue
output = self.tool.outputs[output_name]
yield ExecutionSlice(job_index, param_combination, dataset_collection_elements)
element_identifiers = None
if hasattr(output, "default_identifier_source"):
# Switch the structure for outputs if the output specified a default_identifier_source
collection_type_descriptions = trans.app.dataset_collections_service.collection_type_descriptions
def ensure_implicit_collections_populated(self, history, params):
if not self.collection_info:
return
source_collection = self.collection_info.collections.get(output.default_identifier_source)
if source_collection:
collection_type_description = collection_type_descriptions.for_collection_type(source_collection.collection.collection_type)
_structure = structure.for_dataset_collection(source_collection.collection, collection_type_description=collection_type_description)
if structure.can_match(_structure):
element_identifiers = _structure.element_identifiers_for_outputs(trans, outputs)
if not element_identifiers:
element_identifiers = structure.element_identifiers_for_outputs(trans, outputs)
implicit_collection_info = dict(
implicit_inputs=implicit_inputs,
implicit_output_name=output_name,
outputs=outputs
)
try:
output_collection_name = self.tool.tool_action.get_output_name(
output,
dataset=None,
tool=self.tool,
on_text=on_text,
trans=trans,
history=history,
params=params,
incoming=None,
job_params=None,
)
except Exception:
output_collection_name = "%s across %s" % (self.tool.name, on_text)
child_element_identifiers = element_identifiers["element_identifiers"]
collection_type = element_identifiers["collection_type"]
collection = trans.app.dataset_collections_service.create(
trans=trans,
parent=history,
name=output_collection_name,
element_identifiers=child_element_identifiers,
collection_type=collection_type,
implicit_collection_info=implicit_collection_info,
)
for job in self.successful_jobs:
# TODO: Think through this, may only want this for output
# collections - or we may be already recording data in some
# other way.
if job not in trans.sa_session:
job = trans.sa_session.query(trans.app.model.Job).get(job.id)
job.add_output_dataset_collection(output_name, collection)
collections[output_name] = collection
# Needed to flush the association created just above with
# job.add_output_dataset_collection.
trans.sa_session.flush()
self.implicit_collections = collections
history = history or self.tool.get_default_history_by_trans(self.trans)
if self.invocation_step.is_new:
self.precreate_output_collections(history, params)
else:
collections = {}
for output_assoc in self.invocation_step.output_dataset_collections:
implicit_collection = output_assoc.dataset_collection
assert hasattr(implicit_collection, "history_content_type") # make sure it is an HDCA and not a DC
collections[output_assoc.output_name] = output_assoc.dataset_collection
self.implicit_collections = collections
__all__ = ('execute', )
+4 -1
View File
@@ -195,7 +195,10 @@ class ToolOutputCollectionStructure(object):
if self.structured_like:
collection_prototype = inputs[self.structured_like].collection
else:
collection_prototype = type_registry.prototype(self.collection_type)
collection_type = self.collection_type
assert collection_type
collection_prototype = type_registry.prototype(collection_type)
collection_prototype.collection_type = collection_type
return collection_prototype
@@ -21,6 +21,7 @@ from galaxy.managers.collections_util import (
dictify_dataset_collection_instance,
get_hda_and_element_identifiers
)
from galaxy.managers.jobs import fetch_job_states, summarize_jobs_to_dict
from galaxy.util.json import safe_dumps
from galaxy.util.streamball import StreamBall
from galaxy.web import (
@@ -133,26 +134,104 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary
@expose_api_anonymous
def show(self, trans, id, history_id, **kwd):
"""
show( self, trans, id, history_id, **kwd )
* GET /api/histories/{history_id}/contents/{id}
return detailed information about an HDA within a history
* GET /api/histories/{history_id}/contents/{type}/{id}
return detailed information about an HDA or HDCA within a history
.. note:: Anonymous users are allowed to get their current history contents
:type id: str
:param id: the encoded id of the HDA to return
:param id: the encoded id of the HDA or HDCA to return
:type type: str
:param id: 'dataset' or 'dataset_collection'
:type history_id: str
:param history_id: encoded id string of the HDA's History
:param history_id: encoded id string of the HDA's or HDCA's History
:rtype: dict
:returns: dictionary containing detailed HDA information
:returns: dictionary containing detailed HDA or HDCA information
"""
contents_type = kwd.get('type', 'dataset')
contents_type = self.__get_contents_type(trans, kwd)
if contents_type == 'dataset':
return self.__show_dataset(trans, id, **kwd)
elif contents_type == 'dataset_collection':
return self.__show_dataset_collection(trans, id, history_id, **kwd)
@expose_api_anonymous
def index_jobs_summary(self, trans, history_id, **kwd):
"""
* GET /api/histories/{history_id}/jobs_summary
return detailed information about an HDA or HDCAs jobs
Warning: We allow anyone to fetch job state information about any object they
can guess an encoded ID for - it isn't considered protected data. This keeps
polling IDs as part of state calculation for large histories and collections as
efficient as possible.
:type history_id: str
:param history_id: encoded id string of the HDA's or the HDCA's History
:type ids: str[]
:param ids: the encoded ids of job summary objects to return - if ids
is specified types must also be specified and have same length.
:type types: str[]
:param types: type of object represented by elements in the ids array - either
Job or ImplicitCollectionJob.
:rtype: dict[]
:returns: an array of job summary object dictionaries.
"""
ids = kwd.get("ids", None)
types = kwd.get("types", None)
if ids is None:
assert types is None
# TODO: ...
pass
else:
return self.__handle_unknown_contents_type(trans, contents_type)
ids = util.listify(ids)
types = util.listify(types)
return map(lambda s: self.encode_all_ids(trans, s), fetch_job_states(self.app, trans.sa_session, ids, types))
@expose_api_anonymous
def show_jobs_summary(self, trans, id, history_id, **kwd):
"""
* GET /api/histories/{history_id}/contents/{type}/{id}/jobs_summary
return detailed information about an HDA or HDCAs jobs
Warning: We allow anyone to fetch job state information about any object they
can guess an encoded ID for - it isn't considered protected data. This keeps
polling IDs as part of state calculation for large histories and collections as
efficient as possible.
:type id: str
:param id: the encoded id of the HDA to return
:type history_id: str
:param history_id: encoded id string of the HDA's or the HDCA's History
:rtype: dict
:returns: dictionary containing jobs summary object
"""
contents_type = self.__get_contents_type(trans, kwd)
# At most one of job or implicit_collection_jobs should be found.
job = None
implicit_collection_jobs = None
if contents_type == 'dataset':
hda = self.hda_manager.get_accessible(self.decode_id(id), trans.user)
job = hda.creating_job
elif contents_type == 'dataset_collection':
dataset_collection_instance = self.__get_accessible_collection(trans, id, history_id)
job_source_type = dataset_collection_instance.job_source_type
if job_source_type == "Job":
job = dataset_collection_instance.job
elif job_source_type == "ImplicitCollectionJobs":
implicit_collection_jobs = dataset_collection_instance.implicit_collection_jobs
assert job is None or implicit_collection_jobs is None
return self.encode_all_ids(trans, summarize_jobs_to_dict(trans.sa_session, job or implicit_collection_jobs))
def __get_contents_type(self, trans, kwd):
contents_type = kwd.get('type', 'dataset')
if contents_type not in ['dataset', 'dataset_collection']:
self.__handle_unknown_contents_type(trans, contents_type)
return contents_type
def __show_dataset(self, trans, id, **kwd):
hda = self.hda_manager.get_accessible(self.decode_id(id), trans.user)
@@ -162,18 +241,15 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary
**self._parse_serialization_params(kwd, 'detailed'))
def __show_dataset_collection(self, trans, id, history_id, **kwd):
try:
service = trans.app.dataset_collections_service
dataset_collection_instance = service.get_dataset_collection_instance(
trans=trans,
instance_type='history',
id=id,
)
return self.__collection_dict(trans, dataset_collection_instance, view="element")
except Exception as e:
log.exception("Error in history API at listing dataset collection")
trans.response.status = 500
return {'error': str(e)}
dataset_collection_instance = self.__get_accessible_collection(trans, id, history_id)
return self.__collection_dict(trans, dataset_collection_instance, view="element")
def __get_accessible_collection(self, trans, id, history_id):
return trans.app.dataset_collections_service.get_dataset_collection_instance(
trans=trans,
instance_type="history",
id=id
)
@expose_api_raw_anonymous
def download_dataset_collection(self, trans, id, history_id=None, **kwd):
@@ -188,14 +264,8 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary
:param history_id: encoded id string of the HDCA's History
"""
try:
service = trans.app.dataset_collections_service
dataset_collection_instance = service.get_dataset_collection_instance(
trans=trans,
instance_type='history',
id=id,
)
dataset_collection_instance = self.__get_accessible_collection(trans, id, history_id)
return self.__stream_dataset_collection(trans, dataset_collection_instance)
except Exception as e:
log.exception("Error in API while creating dataset collection archive")
trans.response.status = 500
+9
View File
@@ -363,6 +363,15 @@ def populate_api_routes(webapp, app):
action='download_dataset_collection',
conditions=dict(method=["GET"]))
webapp.mapper.connect("/api/histories/{history_id}/jobs_summary",
action="index_jobs_summary",
controller='history_contents',
conditions=dict(method=["GET"]))
webapp.mapper.connect("/api/histories/{history_id}/contents/{type:%s}s/{id}/jobs_summary" % "|".join(valid_history_contents_types),
action="show_jobs_summary",
controller='history_contents',
conditions=dict(method=["GET"]))
# ---- visualizations registry ---- generic template renderer
# @deprecated: this route should be considered deprecated
webapp.add_route('/visualization/show/{visualization_name}', controller='visualization', action='render', visualization_name=None)
+69 -56
View File
@@ -21,7 +21,7 @@ from galaxy.tools import (
DefaultToolState,
ToolInputsNotReadyException
)
from galaxy.tools.execute import execute
from galaxy.tools.execute import execute, MappingParameters, PartialJobExecution
from galaxy.tools.parameters import (
check_param,
params_to_incoming,
@@ -214,10 +214,15 @@ class WorkflowModule(object):
state.decode(runtime_state, Bunch(inputs=self.get_runtime_inputs()), self.trans.app)
return state
def execute(self, trans, progress, invocation, step):
""" Execute the given workflow step in the given workflow invocation.
def execute(self, trans, progress, invocation_step):
""" Execute the given workflow invocation step.
Use the supplied workflow progress object to track outputs, find
inputs, etc...
inputs, etc....
Return a False if there is additional processing required to
on subsequent workflow scheduling runs, None or True means the workflow
step executed properly.
"""
raise TypeError("Abstract method")
@@ -230,11 +235,19 @@ class WorkflowModule(object):
"""
raise exceptions.RequestParameterInvalidException("Attempting to perform invocation step action on module that does not support actions.")
def recover_mapping(self, step, step_invocations, progress):
def recover_mapping(self, invocation_step, progress):
""" Re-populate progress object with information about connections
from previously executed steps recorded via step_invocations.
from previously executed steps recorded via invocation_steps.
"""
raise TypeError("Abstract method")
outputs = {}
for output_dataset_assoc in invocation_step.output_datasets:
outputs[output_dataset_assoc.output_name] = output_dataset_assoc.dataset
for output_dataset_collection_assoc in invocation_step.output_dataset_collections:
outputs[output_dataset_collection_assoc.output_name] = output_dataset_collection_assoc.dataset_collection
progress.set_step_outputs(invocation_step, outputs, already_persisted=True)
class SubWorkflowModule(WorkflowModule):
@@ -320,11 +333,12 @@ class SubWorkflowModule(WorkflowModule):
def get_content_id(self):
return self.trans.security.encode_id(self.subworkflow.id)
def execute(self, trans, progress, invocation, step):
def execute(self, trans, progress, invocation_step):
""" Execute the given workflow step in the given workflow invocation.
Use the supplied workflow progress object to track outputs, find
inputs, etc...
"""
step = invocation_step.workflow_step
subworkflow_invoker = progress.subworkflow_invoker(trans, step)
subworkflow_invoker.invoke()
subworkflow = subworkflow_invoker.workflow
@@ -334,7 +348,7 @@ class SubWorkflowModule(WorkflowModule):
workflow_output_label = workflow_output.label or "%s:%s" % (step.order_index, workflow_output.output_name)
replacement = subworkflow_progress.get_replacement_workflow_output(workflow_output)
outputs[workflow_output_label] = replacement
progress.set_step_outputs(step, outputs)
progress.set_step_outputs(invocation_step, outputs)
return None
def get_runtime_state(self):
@@ -353,8 +367,10 @@ class InputModule(WorkflowModule):
def get_data_inputs(self):
return []
def execute(self, trans, progress, invocation, step):
job, step_outputs = None, dict(output=step.state.inputs['input'])
def execute(self, trans, progress, invocation_step):
invocation = invocation_step.workflow_invocation
step = invocation_step.workflow_step
step_outputs = dict(output=step.state.inputs['input'])
# Web controller may set copy_inputs_to_history, API controller always sets
# inputs.
@@ -378,11 +394,10 @@ class InputModule(WorkflowModule):
content = next(iter(step_outputs.values()))
if content:
invocation.add_input(content, step.id)
progress.set_outputs_for_input(step, step_outputs)
return job
progress.set_outputs_for_input(invocation_step, step_outputs)
def recover_mapping(self, step, step_invocations, progress):
progress.set_outputs_for_input(step)
def recover_mapping(self, invocation_step, progress):
progress.set_outputs_for_input(invocation_step)
class InputDataModule(InputModule):
@@ -489,10 +504,10 @@ class InputParameterModule(WorkflowModule):
def get_data_inputs(self):
return []
def execute(self, trans, progress, invocation, step):
job, step_outputs = None, dict(output=step.state.inputs['input'])
progress.set_outputs_for_input(step, step_outputs)
return job
def execute(self, trans, progress, invocation_step):
step = invocation_step.workflow_step
step_outputs = dict(output=step.state.inputs['input'])
progress.set_outputs_for_input(invocation_step, step_outputs)
class PauseModule(WorkflowModule):
@@ -520,18 +535,18 @@ class PauseModule(WorkflowModule):
state.inputs = dict()
return state
def execute(self, trans, progress, invocation, step):
def execute(self, trans, progress, invocation_step):
step = invocation_step.workflow_step
progress.mark_step_outputs_delayed(step, why="executing pause step")
return None
def recover_mapping(self, step, step_invocations, progress):
if step_invocations:
step_invocation = step_invocations[0]
action = step_invocation.action
def recover_mapping(self, invocation_step, progress):
if invocation_step:
step = invocation_step.workflow_step
action = invocation_step.action
if action:
connection = step.input_connections_by_name["input"][0]
replacement = progress.replacement_for_connection(connection)
progress.set_step_outputs(step, {'output': replacement})
progress.set_step_outputs(invocation_step, {'output': replacement})
return
elif action is False:
raise CancelWorkflowEvaluation()
@@ -790,7 +805,9 @@ class ToolModule(WorkflowModule):
else:
raise ToolMissingException("Tool %s missing. Cannot recover runtime state." % self.tool_id)
def execute(self, trans, progress, invocation, step):
def execute(self, trans, progress, invocation_step):
invocation = invocation_step.workflow_invocation
step = invocation_step.workflow_step
tool = trans.app.toolbox.get_tool(step.tool_id, tool_version=step.tool_version)
tool_state = step.state
# Not strictly needed - but keep Tool state clean by stripping runtime
@@ -855,57 +872,53 @@ class ToolModule(WorkflowModule):
param_combinations.append(execution_state.inputs)
complete = False
try:
mapping_params = MappingParameters(tool_state.inputs, param_combinations)
max_num_jobs = progress.maximum_jobs_to_schedule_or_none
execution_tracker = execute(
trans=self.trans,
tool=tool,
param_combinations=param_combinations,
mapping_params=mapping_params,
history=invocation.history,
collection_info=collection_info,
workflow_invocation_uuid=invocation.uuid.hex
workflow_invocation_uuid=invocation.uuid.hex,
invocation_step=invocation_step,
max_num_jobs=max_num_jobs,
job_callback=lambda job: self._handle_post_job_actions(step, job, invocation.replacement_dict),
)
complete = True
except PartialJobExecution as pje:
execution_tracker = pje.execution_tracker
except ToolInputsNotReadyException:
delayed_why = "tool [%s] inputs are not ready, this special tool requires inputs to be ready" % tool.id
raise DelayedWorkflowEvaluation(why=delayed_why)
progress.record_executed_job_count(len(execution_tracker.successful_jobs))
if collection_info:
step_outputs = dict(execution_tracker.implicit_collections)
else:
step_outputs = dict(execution_tracker.output_datasets)
step_outputs.update(execution_tracker.output_collections)
progress.set_step_outputs(step, step_outputs)
jobs = execution_tracker.successful_jobs
for job in jobs:
self._handle_post_job_actions(step, job, invocation.replacement_dict)
progress.set_step_outputs(invocation_step, step_outputs, already_persisted=not invocation_step.is_new)
if execution_tracker.execution_errors:
failed_count = len(execution_tracker.execution_errors)
success_count = len(execution_tracker.successful_jobs)
all_count = failed_count + success_count
message = "Failed to create %d out of %s job(s) for workflow step." % (failed_count, all_count)
message = "Failed to create one or more job(s) for workflow step."
raise Exception(message)
return jobs
def recover_mapping(self, step, step_invocations, progress):
# Grab a job representing this invocation - for normal workflows
# there will be just one job but if this step was mapped over there
# may be many.
job_0 = step_invocations[0].job
return complete
def recover_mapping(self, invocation_step, progress):
outputs = {}
for job_output in job_0.output_datasets:
replacement_name = job_output.name
replacement_value = job_output.dataset
# If was a mapping step, grab the output mapped collection for
# replacement instead.
if replacement_value.hidden_beneath_collection_instance:
replacement_value = replacement_value.hidden_beneath_collection_instance
outputs[replacement_name] = replacement_value
for job_output_collection in job_0.output_dataset_collection_instances:
replacement_name = job_output_collection.name
replacement_value = job_output_collection.dataset_collection_instance
outputs[replacement_name] = replacement_value
progress.set_step_outputs(step, outputs)
for output_dataset_assoc in invocation_step.output_datasets:
outputs[output_dataset_assoc.output_name] = output_dataset_assoc.dataset
for output_dataset_collection_assoc in invocation_step.output_dataset_collections:
outputs[output_dataset_collection_assoc.output_name] = output_dataset_collection_assoc.dataset_collection
progress.set_step_outputs(invocation_step, outputs)
def _find_collections_to_match(self, tool, progress, step):
collections_to_match = matching.CollectionsToMatch()
+85 -31
View File
@@ -1,7 +1,7 @@
import logging
import uuid
from galaxy import model, util
from galaxy import model
from galaxy.util import ExecutionTimer
from galaxy.util.odict import odict
from galaxy.workflow import modules
@@ -140,12 +140,18 @@ class WorkflowInvoker(object):
module_injector = modules.WorkflowModuleInjector(trans)
if progress is None:
progress = WorkflowProgress(self.workflow_invocation, workflow_run_config.inputs, module_injector)
progress = WorkflowProgress(
self.workflow_invocation,
workflow_run_config.inputs,
module_injector,
jobs_per_scheduling_iteration=getattr(trans.app.config, "maximum_workflow_jobs_per_scheduling_iteration", -1),
)
self.progress = progress
def invoke(self):
workflow_invocation = self.workflow_invocation
maximum_duration = getattr(self.trans.app.config, "maximum_workflow_invocation_duration", -1)
config = self.trans.app.config
maximum_duration = getattr(config, "maximum_workflow_invocation_duration", -1)
if maximum_duration > 0 and workflow_invocation.seconds_since_created > maximum_duration:
log.debug("Workflow invocation [%s] exceeded maximum number of seconds allowed for scheduling [%s], failing." % (workflow_invocation.id, maximum_duration))
workflow_invocation.state = model.WorkflowInvocation.states.FAILED
@@ -162,24 +168,27 @@ class WorkflowInvoker(object):
remaining_steps = self.progress.remaining_steps()
delayed_steps = False
for step in remaining_steps:
for (step, workflow_invocation_step) in remaining_steps:
step_delayed = False
step_timer = ExecutionTimer()
jobs = None
try:
self.__check_implicitly_dependent_steps(step)
# TODO: step may fail to invoke, do something about that.
jobs = self._invoke_step(step)
for job in (util.listify(jobs) or [None]):
# Record invocation
if not workflow_invocation_step:
workflow_invocation_step = model.WorkflowInvocationStep()
workflow_invocation_step.workflow_invocation = workflow_invocation
workflow_invocation_step.workflow_step = step
# Job may not be generated in this thread if bursting is enabled
# https://github.com/galaxyproject/galaxy/issues/2259
if job:
workflow_invocation_step.job_id = job.id
workflow_invocation_step.state = 'new'
workflow_invocation.steps.append(workflow_invocation_step)
incomplete_or_none = self._invoke_step(workflow_invocation_step)
if incomplete_or_none is False:
step_delayed = delayed_steps = True
workflow_invocation_step.state = 'ready'
self.progress.mark_step_outputs_delayed(step, why="Not all jobs scheduled for state.")
else:
workflow_invocation_step.state = 'scheduled'
except modules.DelayedWorkflowEvaluation as de:
step_delayed = delayed_steps = True
self.progress.mark_step_outputs_delayed(step, why=de.why)
@@ -218,15 +227,19 @@ class WorkflowInvoker(object):
self.__check_implicitly_dependent_step(output_id)
def __check_implicitly_dependent_step(self, output_id):
step_invocations = self.workflow_invocation.step_invocations_for_step_id(output_id)
step_invocation = self.workflow_invocation.step_invocation_for_step_id(output_id)
# No steps created yet - have to delay evaluation.
if not step_invocations:
if not step_invocation:
delayed_why = "depends on step [%s] but that step has not been invoked yet" % output_id
raise modules.DelayedWorkflowEvaluation(why=delayed_why)
for step_invocation in step_invocations:
job = step_invocation.job
if step_invocation.state != 'scheduled':
delayed_why = "depends on step [%s] job has not finished scheduling yet" % output_id
raise modules.DelayedWorkflowEvaluation(delayed_why)
for job_assoc in step_invocation.jobs:
job = job_assoc.job
if job:
# At least one job in incomplete.
if not job.finished:
@@ -241,9 +254,9 @@ class WorkflowInvoker(object):
# pause steps.
pass
def _invoke_step(self, step):
jobs = step.module.execute(self.trans, self.progress, self.workflow_invocation, step)
return jobs
def _invoke_step(self, invocation_step):
incomplete_or_none = invocation_step.workflow_step.module.execute(self.trans, self.progress, invocation_step)
return incomplete_or_none
STEP_OUTPUT_DELAYED = object()
@@ -251,16 +264,31 @@ STEP_OUTPUT_DELAYED = object()
class WorkflowProgress(object):
def __init__(self, workflow_invocation, inputs_by_step_id, module_injector):
def __init__(self, workflow_invocation, inputs_by_step_id, module_injector, jobs_per_scheduling_iteration=-1):
self.outputs = odict()
self.module_injector = module_injector
self.workflow_invocation = workflow_invocation
self.inputs_by_step_id = inputs_by_step_id
self.jobs_per_scheduling_iteration = jobs_per_scheduling_iteration
self.jobs_scheduled_this_iteration = 0
@property
def maximum_jobs_to_schedule_or_none(self):
if self.jobs_per_scheduling_iteration > 0:
return self.jobs_per_scheduling_iteration - self.jobs_scheduled_this_iteration
else:
return None
def record_executed_job_count(self, job_count):
self.jobs_scheduled_this_iteration += job_count
def remaining_steps(self):
# Previously computed and persisted step states.
step_states = self.workflow_invocation.step_states_by_step_id()
steps = self.workflow_invocation.workflow.steps
# TODO: Wouldn't a generator be much better here so we don't have to reason about
# steps we are no where near ready to schedule?
remaining_steps = []
step_invocations_by_id = self.workflow_invocation.step_invocations_by_step_id()
for step in steps:
@@ -274,11 +302,11 @@ class WorkflowProgress(object):
runtime_state = step_states[step_id].value
step.state = step.module.decode_runtime_state(runtime_state)
invocation_steps = step_invocations_by_id.get(step_id, None)
if invocation_steps:
self._recover_mapping(step, invocation_steps)
invocation_step = step_invocations_by_id.get(step_id, None)
if invocation_step and invocation_step.state == 'scheduled':
self._recover_mapping(invocation_step)
else:
remaining_steps.append(step)
remaining_steps.append((step, invocation_step))
return remaining_steps
def replacement_for_tool_input(self, step, input, prefixed_name):
@@ -340,7 +368,9 @@ class WorkflowProgress(object):
output_name = workflow_output.output_name
return self.outputs[step.id][output_name]
def set_outputs_for_input(self, step, outputs=None):
def set_outputs_for_input(self, invocation_step, outputs=None):
step = invocation_step.workflow_step
if outputs is None:
outputs = {}
@@ -352,10 +382,34 @@ class WorkflowProgress(object):
raise ValueError(message)
outputs['output'] = self.inputs_by_step_id[step_id]
self.set_step_outputs(step, outputs)
self.set_step_outputs(invocation_step, outputs)
def set_step_outputs(self, step, outputs):
def set_step_outputs(self, invocation_step, outputs, already_persisted=False):
step = invocation_step.workflow_step
self.outputs[step.id] = outputs
if not already_persisted:
for output_name, output_object in outputs.items():
if hasattr(output_object, "history_content_type"):
invocation_step.add_output(output_name, output_object)
else:
# This is a problem, this non-data, non-collection output
# won't be recovered on a subsequent workflow scheduling
# iteration. This seems to have been a pre-existing problem
# prior to #4584 though.
pass
for workflow_output in step.workflow_outputs:
output_name = workflow_output.output_name
if output_name not in outputs:
raise KeyError("Failed to find [%s] in step outputs [%s]" % (output_name, outputs))
output = outputs[output_name]
self._record_workflow_output(
step,
workflow_output,
output=output,
)
def _record_workflow_output(self, step, workflow_output, output):
self.workflow_invocation.add_output(workflow_output, step, output)
def mark_step_outputs_delayed(self, step, why=None):
if why:
@@ -414,11 +468,11 @@ class WorkflowProgress(object):
self.module_injector,
)
def _recover_mapping(self, step, step_invocations):
def _recover_mapping(self, step_invocation):
try:
step.module.recover_mapping(step, step_invocations, self)
step_invocation.workflow_step.module.recover_mapping(step_invocation, self)
except modules.DelayedWorkflowEvaluation as de:
self.mark_step_outputs_delayed(step, de.why)
self.mark_step_outputs_delayed(step_invocation.workflow_step, de.why)
__all__ = ('invoke', 'WorkflowRunConfig')
+1
View File
@@ -23,6 +23,7 @@ from six import string_types
from sqlalchemy import * # noqa
from sqlalchemy.orm import * # noqa
from sqlalchemy.exc import * # noqa
from sqlalchemy.sql import label # noqa
sys.path.insert(1, os.path.abspath(os.path.join(os.path.dirname(__file__), os.pardir, 'lib')))
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+1 -1
View File
@@ -1 +1 @@
define("mvc/history/hdca-li",["exports","mvc/dataset/states","mvc/collection/collection-li","mvc/collection/collection-view","mvc/base-mvc","mvc/history/history-item-li","utils/localization"],function(e,t,i,n,s,l,a){"use strict";function o(e){return e&&e.__esModule?e:{default:e}}Object.defineProperty(e,"__esModule",{value:!0});var c=o(t),r=o(i),d=o(n),u=(o(s),o(l)),p=o(a),h=r.default.DCListItemView,m=h.extend({className:h.prototype.className+" history-content",_setUpListeners:function(){h.prototype._setUpListeners.call(this),this.listenTo(this.model,{"change:tags change:populated change:visible":function(e,t){this.render()}})},_getFoldoutPanelClass:function(){var e=this.model.get("collection_type");switch(e){case"list":return d.default.ListCollectionView;case"paired":return d.default.PairCollectionView;case"list:paired":return d.default.ListOfPairsCollectionView;case"list:list":return d.default.ListOfListsCollectionView}throw new TypeError("Unknown collection_type: "+e)},_swapNewRender:function(e){h.prototype._swapNewRender.call(this,e);var t=this.model.get("populated")?c.default.OK:c.default.RUNNING;return this.$el.addClass("state-"+t),this.$el},toString:function(){return"HDCAListItemView("+(this.model?""+this.model:"(no model)")+")"}});m.prototype.templates=function(){var e=_.extend({},h.prototype.templates.warnings,{hidden:function(e){e.visible||(0,p.default)("This collection has been hidden")}});return _.extend({},h.prototype.templates,{warnings:e,titleBar:function(e){return'\n <div class="title-bar clear" tabindex="0">\n <span class="state-icon"></span>\n <div class="title">\n <span class="hid">'+e.hid+'</span>\n <span class="name">'+_.escape(e.name)+'</span>\n </div>\n <div class="subtitle"></div>\n '+u.default.nametagTemplate(e)+"\n </div>\n "}})}(),e.default={HDCAListItemView:m}});
define("mvc/history/hdca-li",["exports","mvc/dataset/states","mvc/collection/collection-li","mvc/collection/collection-view","mvc/base-mvc","mvc/history/history-item-li","utils/localization"],function(e,t,s,i,n,a,r){"use strict";function o(e){return e&&e.__esModule?e:{default:e}}Object.defineProperty(e,"__esModule",{value:!0});var l=o(t),d=o(s),c=o(i),p=(o(n),o(a)),u=o(r),h=d.default.DCListItemView,m=h.extend({className:h.prototype.className+" history-content",_setUpListeners:function(){var e=this;h.prototype._setUpListeners.call(this);var t=function(t,s){e.render()};this.model.jobStatesSummary&&this.listenTo(this.model.jobStatesSummary,"change",t),this.listenTo(this.model,{"change:tags change:visible change:state":t})},_getFoldoutPanelClass:function(){var e=this.model.get("collection_type");switch(e){case"list":return c.default.ListCollectionView;case"paired":return c.default.PairCollectionView;case"list:paired":return c.default.ListOfPairsCollectionView;case"list:list":return c.default.ListOfListsCollectionView}throw new TypeError("Unknown collection_type: "+e)},_swapNewRender:function(e){h.prototype._swapNewRender.call(this,e);var t,s=this.model.jobStatesSummary;t=s?s.new()?"loading":s.errored()?"error":s.terminal()?"ok":s.running()?"running":"queued":this.model.get("job_source_id")?"loading":this.model.get("populated_state")?l.default.OK:l.default.RUNNING,this.$el.addClass("state-"+t);var i=this.stateDescription();return this.$(".state-description").html(i),this.$el},stateDescription:function(){var e,t=this.model,s=t.get("element_count"),i=t.get("job_source_type"),n=this.model.get("collection_type");e="list"==n?"list":"paired"==n?"dataset pair":"list:paired"==n?"list of pairs":"nested list";var a="";1==s?a=" with 1 item":s&&(a=" with "+s+" items");var r=t.jobStatesSummary,o=""+e+a;if(i&&"Job"!=i){if(r&&r.hasDetails()){var l=r.new(),d=l?null:r.jobCount();if(l)return'\n <div class="progress state-progress">\n <span class="note">Creating jobs.<span class="blinking">..</span></span>\n <div class="progress-bar info" style="width:100%">\n </div>';if(r.errored())return"a "+e+" with "+r.numInError()+" / "+d+" jobs in error";if(r.terminal())return"a "+o;var c=r.states().running||0,p=(r.states().ok||0)/(1*d),u=c/(1*d),h=1-p-u;return'\n <div class="progress state-progress">\n <span class="note">'+(d&&d>1?d+" jobs":"a job")+" generating a "+e+'</span>\n <div class="progress-bar ok" style="width:'+100*p+'%"></div>\n <div class="progress-bar running" style="width:'+100*u+'%"></div>\n <div class="progress-bar new" style="width:'+100*h+'%">\n </div>'}return'\n <div class="progress state-progress">\n <span class="note">Loading job data for '+e+'.<span class="blinking">..</span></span>\n <div class="progress-bar info" style="width:100%">\n </div>'}return"a "+o},toString:function(){return"HDCAListItemView("+(this.model?""+this.model:"(no model)")+")"}});m.prototype.templates=function(){var e=_.extend({},h.prototype.templates.warnings,{hidden:function(e){e.visible||(0,u.default)("This collection has been hidden")}});return _.extend({},h.prototype.templates,{warnings:e,titleBar:function(e){return'\n <div class="title-bar clear" tabindex="0">\n <span class="state-icon"></span>\n <div class="title">\n <span class="hid">'+e.hid+'</span>\n <span class="name">'+_.escape(e.name)+'</span>\n </div>\n <div class="state-description">\n </div>\n '+p.default.nametagTemplate(e)+"\n </div>\n "}})}(),e.default={HDCAListItemView:m}});
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1 @@
define("mvc/history/job-states-model",["exports","libs/backbone","utils/ajax-queue"],function(t,e,n){"use strict";Object.defineProperty(t,"__esModule",{value:!0});var i=function(t){if(t&&t.__esModule)return t;var e={};if(null!=t)for(var n in t)Object.prototype.hasOwnProperty.call(t,n)&&(e[n]=t[n]);return e.default=t,e}(e),r=function(t){return t&&t.__esModule?t:{default:t}}(n),u=["new","queued","running"],o=["error","deleted"],a=i.Model.extend({url:function(){return Galaxy.root+"api/histories/"+this.attributes.history_id+"/contents/dataset_collections/"+this.attributes.collection_id+"/jobs_summary"},hasDetails:function(){return this.has("populated_state")},new:function(){return!this.hasDetails()||"new"==this.get("populated_state")},errored:function(){return"error"===this.get("populated_state")||this.anyWithStates(o)},states:function(){return this.get("states")||{}},anyWithState:function(t){return(this.states()[t]||0)>0},anyWithStates:function(t){var e=this.states();for(var n in t)if((e[t[n]]||0)>0)return!0;return!1},numWithStates:function(t){var e=this.states(),n=0;for(var i in t)n+=e[t[i]]||0;return n},numInError:function(){return this.numWithStates(o)},running:function(){return this.anyWithState("running")},terminal:function(){return!this.new()&&!this.anyWithStates(u)},jobCount:function(){var t=this.states(),e=0;for(var n in t)e+=t[n];return e},toString:function(){return"JobStatesSummary(id="+this.get("id")+")"}}),s=i.Collection.extend({model:a,initialize:function(){this.updateTimeoutId=null,this.active=!0},url:function(){var t=this.models.filter(function(t){return!t.terminal()}),e=t.map(function(t){return t.get("id")}).join(","),n=t.map(function(t){return t.get("model")}).join(",");return Galaxy.root+"api/histories/"+this.historyId+"/jobs_summary?ids="+e+"&types="+n},monitor:function(){var t=this;if(this.clearUpdateTimeout(),this.active){var e=function(){t.updateTimeoutId=setTimeout(function(){t.monitor()},2e3)},n=this.models.filter(function(t){return!t.terminal()});if(n.length,!1){var i=n.map(function(t){return function(){return t.fetch()}});return new r.default.AjaxQueue(i).done(e)}n.length>0?this.fetch({remove:!1}).done(e):e()}},clearUpdateTimeout:function(){this.updateTimeoutId&&(clearTimeout(this.updateTimeoutId),this.updateTimeoutId=null)},toString:function(){return"JobStatesSummaryCollection()"}});t.default={JobStatesSummary:a,JobStatesSummaryCollection:s,FETCH_STATE_ON_ADD:!1}});
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+2 -2
View File
@@ -87,7 +87,7 @@ class DatasetCollectionApiTestCase(api.ApiTestCase):
pair_1_element = returned_collections[0]
self._assert_has_keys(pair_1_element, "element_index")
pair_1_object = pair_1_element["object"]
self._assert_has_keys(pair_1_object, "collection_type", "elements")
self._assert_has_keys(pair_1_object, "collection_type", "elements", "element_count")
self.assertEquals(pair_1_object["collection_type"], "paired")
self.assertEquals(pair_1_object["populated"], True)
pair_elements = pair_1_object["elements"]
@@ -195,7 +195,7 @@ class DatasetCollectionApiTestCase(api.ApiTestCase):
def _check_create_response(self, create_response):
self._assert_status_code_is(create_response, 200)
dataset_collection = create_response.json()
self._assert_has_keys(dataset_collection, "elements", "url", "name", "collection_type")
self._assert_has_keys(dataset_collection, "elements", "url", "name", "collection_type", "element_count")
return dataset_collection
def _download_dataset_collection(self, history_id, hdca_id):
+36 -1
View File
@@ -6,9 +6,11 @@ from requests import delete, put
from base import api # noqa: I100
from base.populators import ( # noqa: I100
DatasetPopulator,
DatasetCollectionPopulator,
LibraryPopulator,
TestsDatasets
skip_without_tool,
TestsDatasets,
)
@@ -18,6 +20,7 @@ class HistoryContentsApiTestCase(api.ApiTestCase, TestsDatasets):
def setUp(self):
super(HistoryContentsApiTestCase, self).setUp()
self.history_id = self._new_history()
self.dataset_populator = DatasetPopulator(self.galaxy_interactor)
self.dataset_collection_populator = DatasetCollectionPopulator(self.galaxy_interactor)
self.library_populator = LibraryPopulator(self)
@@ -177,6 +180,38 @@ class HistoryContentsApiTestCase(api.ApiTestCase, TestsDatasets):
dataset_collection = show_response.json()
assert dataset_collection["deleted"]
@skip_without_tool("collection_creates_list")
def test_jobs_summary_simple_hdca(self):
create_response = self.dataset_collection_populator.create_list_in_history(self.history_id, contents=["a\nb\nc\nd", "e\nf\ng\nh"])
hdca_id = create_response.json()["id"]
run = self.dataset_populator.run_collection_creates_list(self.history_id, hdca_id)
collections = run['output_collections']
collection = collections[0]
jobs_summary_url = "histories/%s/contents/dataset_collections/%s/jobs_summary" % (self.history_id, collection["id"])
jobs_summary_response = self._get(jobs_summary_url)
self._assert_status_code_is(jobs_summary_response, 200)
jobs_summary = jobs_summary_response.json()
self._assert_has_keys(jobs_summary, "populated_state", "states")
@skip_without_tool("cat1")
def test_jobs_summary_implicit_hdca(self):
create_response = self.dataset_collection_populator.create_pair_in_history(self.history_id, contents=["123", "456"])
hdca_id = create_response.json()["id"]
inputs = {
"input1": {'batch': True, 'values': [{'src': 'hdca', 'id': hdca_id}]},
}
run = self.dataset_populator.run_tool("cat1", inputs=inputs, history_id=self.history_id)
self.dataset_populator.wait_for_history_jobs(self.history_id)
collections = run['implicit_collections']
collection = collections[0]
jobs_summary_url = "histories/%s/contents/dataset_collections/%s/jobs_summary" % (self.history_id, collection["id"])
jobs_summary_response = self._get(jobs_summary_url)
self._assert_status_code_is(jobs_summary_response, 200)
jobs_summary = jobs_summary_response.json()
self._assert_has_keys(jobs_summary, "populated_state", "states")
states = jobs_summary["states"]
assert states.get("ok") == 2, states
def test_dataset_collection_hide_originals(self):
payload = self.dataset_collection_populator.create_pair_payload(
self.history_id,
+67 -20
View File
@@ -206,11 +206,7 @@ class ToolsTestCase(api.ApiTestCase):
with self.dataset_populator.test_history() as history_id:
history_id = self.dataset_populator.new_history()
ok_hdca_id = self.dataset_collection_populator.create_list_in_history(history_id, contents=["0", "1", "0", "1"]).json()["id"]
exit_code_inputs = {
"input": {'batch': True, 'values': [{"src": "hdca", "id": ok_hdca_id}]},
}
response = self._run("exit_code_from_file", history_id, exit_code_inputs, assert_ok=False).json()
self.dataset_populator.wait_for_history(history_id, assert_ok=False)
response = self.dataset_populator.run_exit_code_from_file(history_id, ok_hdca_id)
mixed_implicit_collections = response["implicit_collections"]
self.assertEquals(len(mixed_implicit_collections), 1)
@@ -433,20 +429,15 @@ class ToolsTestCase(api.ApiTestCase):
@skip_without_tool("collection_creates_list")
def test_list_collection_output(self):
history_id = self.dataset_populator.new_history()
create_response = self.dataset_collection_populator.create_list_in_history(history_id, contents=["a\nb\nc\nd", "e\nf\ng\nh"])
hdca_id = create_response.json()["id"]
inputs = {
"input1": {"src": "hdca", "id": hdca_id},
}
# TODO: real problem here - shouldn't have to have this wait.
self.dataset_populator.wait_for_history(history_id, assert_ok=True)
create = self._run("collection_creates_list", history_id, inputs, assert_ok=True)
output_collection = self._assert_one_job_one_collection_run(create)
element0, element1 = self._assert_elements_are(output_collection, "data1", "data2")
self.dataset_populator.wait_for_history(history_id, assert_ok=True)
self._verify_element(history_id, element0, contents="identifier is data1\n", file_ext="txt")
self._verify_element(history_id, element1, contents="identifier is data2\n", file_ext="txt")
with self.dataset_populator.test_history() as history_id:
create_response = self.dataset_collection_populator.create_list_in_history(history_id, contents=["a\nb\nc\nd", "e\nf\ng\nh"])
hdca_id = create_response.json()["id"]
create = self.dataset_populator.run_collection_creates_list(history_id, hdca_id)
output_collection = self._assert_one_job_one_collection_run(create)
element0, element1 = self._assert_elements_are(output_collection, "data1", "data2")
self.dataset_populator.wait_for_history(history_id, assert_ok=True)
self._verify_element(history_id, element0, contents="identifier is data1\n", file_ext="txt")
self._verify_element(history_id, element1, contents="identifier is data2\n", file_ext="txt")
@skip_without_tool("collection_creates_list_2")
def test_list_collection_output_format_source(self):
@@ -681,6 +672,24 @@ class ToolsTestCase(api.ApiTestCase):
}
self._run_and_check_simple_collection_mapping(history_id, inputs)
@skip_without_tool("cat1")
def test_map_over_empty_collection(self):
with self.dataset_populator.test_history() as history_id:
hdca_id = self.dataset_collection_populator.create_list_in_history(history_id, contents=[]).json()['id']
inputs = {
"input1": {'batch': True, 'values': [{'src': 'hdca', 'id': hdca_id}]},
}
create = self._run_cat1(history_id, inputs=inputs, assert_ok=True)
outputs = create['outputs']
jobs = create['jobs']
implicit_collections = create['implicit_collections']
self.assertEquals(len(jobs), 0)
self.assertEquals(len(outputs), 0)
self.assertEquals(len(implicit_collections), 1)
empty_output = implicit_collections[0]
assert empty_output["name"] == "Concatenate datasets on collection 1", empty_output
@skip_without_tool("output_action_change_format")
def test_map_over_with_output_format_actions(self):
for use_action in ["do", "dont"]:
@@ -704,6 +713,38 @@ class ToolsTestCase(api.ApiTestCase):
assert output1_details["file_ext"] == "txt" if (use_action == "do") else "data"
assert output2_details["file_ext"] == "txt" if (use_action == "do") else "data"
@skip_without_tool("output_filter_with_input")
def test_map_over_with_output_filter_no_filtering(self):
with self.dataset_populator.test_history() as history_id:
hdca_id = self.dataset_collection_populator.create_list_in_history(history_id).json()["id"]
inputs = {
"input_1": {'batch': True, 'values': [{'src': 'hdca', 'id': hdca_id}]},
"produce_out_1": "true",
"filter_text_1": "foo",
}
create = self._run('output_filter_with_input', history_id, inputs).json()
jobs = create['jobs']
implicit_collections = create['implicit_collections']
self.assertEquals(len(jobs), 3)
self.assertEquals(len(implicit_collections), 3)
self._check_implicit_collection_populated(create)
@skip_without_tool("output_filter_with_input")
def test_map_over_with_output_filter_one_filtered(self):
with self.dataset_populator.test_history() as history_id:
hdca_id = self.dataset_collection_populator.create_list_in_history(history_id).json()["id"]
inputs = {
"input_1": {'batch': True, 'values': [{'src': 'hdca', 'id': hdca_id}]},
"produce_out_1": "true",
"filter_text_1": "bar",
}
create = self._run('output_filter_with_input', history_id, inputs).json()
jobs = create['jobs']
implicit_collections = create['implicit_collections']
self.assertEquals(len(jobs), 3)
self.assertEquals(len(implicit_collections), 2)
self._check_implicit_collection_populated(create)
@skip_without_tool("Cut1")
def test_map_over_with_complex_output_actions(self):
history_id = self.dataset_populator.new_history()
@@ -951,7 +992,7 @@ class ToolsTestCase(api.ApiTestCase):
self.assertEquals(len(jobs), 2)
self.assertEquals(len(implicit_collections), 1)
implicit_collection = implicit_collections[0]
assert implicit_collection["collection_type"] == "list:paired", implicit_collection
assert implicit_collection["collection_type"] == "list:paired", implicit_collection["collection_type"]
outer_elements = implicit_collection["elements"]
assert len(outer_elements) == 2
@@ -1297,6 +1338,12 @@ class ToolsTestCase(api.ApiTestCase):
assert output1_content.strip() == "123\n456\nxxx", output1_content
assert output2_content.strip() == "789\n0ab\nyyy", output2_content
def _check_implicit_collection_populated(self, run_response):
implicit_collections = run_response["implicit_collections"]
assert implicit_collections
for implicit_collection in implicit_collections:
assert implicit_collection["populated_state"] == "ok"
def _cat1_outputs(self, history_id, inputs):
return self._run_outputs(self._run_cat1(history_id, inputs))
+6 -2
View File
@@ -5,7 +5,7 @@ import operator
from collections import namedtuple
from json import dumps, loads
from base.populators import skip_without_tool
from base.populators import skip_without_tool, summarize_instance_history_on_error
from .test_workflows import BaseWorkflowsApiTestCase
@@ -17,6 +17,7 @@ class WorkflowExtractionApiTestCase(BaseWorkflowsApiTestCase):
self.history_id = self.dataset_populator.new_history()
@skip_without_tool("cat1")
@summarize_instance_history_on_error
def test_extract_from_history(self):
# Run the simple test workflow and extract it back out from history
cat1_job_id = self.__setup_and_run_cat1_workflow(history_id=self.history_id)
@@ -29,6 +30,7 @@ class WorkflowExtractionApiTestCase(BaseWorkflowsApiTestCase):
self.assertEqual(downloaded_workflow["name"], "test import from history")
self.__assert_looks_like_cat1_example_workflow(downloaded_workflow)
@summarize_instance_history_on_error
def test_extract_with_copied_inputs(self):
old_history_id = self.dataset_populator.new_history()
# Run the simple test workflow and extract it back out from history
@@ -54,6 +56,7 @@ class WorkflowExtractionApiTestCase(BaseWorkflowsApiTestCase):
self.__assert_looks_like_cat1_example_workflow(downloaded_workflow)
@skip_without_tool("random_lines1")
@summarize_instance_history_on_error
def test_extract_mapping_workflow_from_history(self):
hdca, job_id1, job_id2 = self.__run_random_lines_mapped_over_pair(self.history_id)
downloaded_workflow = self._extract_and_download_workflow(
@@ -232,6 +235,7 @@ test_data:
)
@skip_without_tool("collection_creates_pair")
@summarize_instance_history_on_error
def test_extract_with_mapped_output_collections(self):
jobs_summary = self._run_jobs("""
class: GalaxyWorkflow
@@ -463,7 +467,7 @@ test_data:
disconnected_inputs.append(value)
if disconnected_inputs:
template = "%d step(s_ disconnected in extracted workflow - disconnectect steps are %s - workflow is %s"
template = "%d steps disconnected in extracted workflow - disconnectect steps are %s - workflow is %s"
message = template % (len(disconnected_inputs), disconnected_inputs, workflow)
raise AssertionError(message)
+416 -36
View File
@@ -24,6 +24,9 @@ SIMPLE_NESTED_WORKFLOW_YAML = """
class: GalaxyWorkflow
inputs:
- id: outer_input
outputs:
- id: outer_output
source: second_cat#out_file1
steps:
- tool_id: cat1
label: first_cat
@@ -58,11 +61,6 @@ steps:
queries:
- input2:
$link: nested_workflow#workflow_output
test_data:
outer_input:
value: 1.bed
type: File
"""
@@ -734,8 +732,8 @@ steps:
def test_workflow_run_dynamic_output_collections_2(self):
# A more advanced output collection workflow, testing regression of
# https://github.com/galaxyproject/galaxy/issues/776
history_id = self.dataset_populator.new_history()
workflow_id = self._upload_yaml_workflow("""
with self.dataset_populator.test_history() as history_id:
workflow_id = self._upload_yaml_workflow("""
class: GalaxyWorkflow
steps:
- label: test_input_1
@@ -759,19 +757,21 @@ steps:
- input2:
$link: split_up#split_output
""")
hda1 = self.dataset_populator.new_dataset(history_id, content="samp1\t10.0\nsamp2\t20.0\n")
hda2 = self.dataset_populator.new_dataset(history_id, content="samp1\t20.0\nsamp2\t40.0\n")
hda3 = self.dataset_populator.new_dataset(history_id, content="samp1\t30.0\nsamp2\t60.0\n")
self.dataset_populator.wait_for_history(history_id, assert_ok=True)
inputs = {
'0': self._ds_entry(hda1),
'1': self._ds_entry(hda2),
'2': self._ds_entry(hda3),
}
invocation_id = self.__invoke_workflow(history_id, workflow_id, inputs)
self.wait_for_invocation_and_jobs(history_id, workflow_id, invocation_id)
content = self.dataset_populator.get_history_dataset_content(history_id, hid=7)
self.assertEqual(content.strip(), "samp1\t10.0\nsamp2\t20.0")
hda1 = self.dataset_populator.new_dataset(history_id, content="samp1\t10.0\nsamp2\t20.0\n")
hda2 = self.dataset_populator.new_dataset(history_id, content="samp1\t20.0\nsamp2\t40.0\n")
hda3 = self.dataset_populator.new_dataset(history_id, content="samp1\t30.0\nsamp2\t60.0\n")
self.dataset_populator.wait_for_history(history_id, assert_ok=True)
inputs = {
'0': self._ds_entry(hda1),
'1': self._ds_entry(hda2),
'2': self._ds_entry(hda3),
}
invocation_id = self.__invoke_workflow(history_id, workflow_id, inputs)
self.wait_for_invocation_and_jobs(history_id, workflow_id, invocation_id)
collection_details = self.dataset_populator.get_history_collection_details(history_id, hid=7)
assert collection_details["populated_state"] == "ok"
content = self.dataset_populator.get_history_dataset_content(history_id, hid=11)
self.assertEqual(content.strip(), "samp1\t10.0\nsamp2\t20.0")
@skip_without_tool("collection_split_on_column")
def test_workflow_run_dynamic_output_collections_3(self):
@@ -861,7 +861,14 @@ test_data:
def test_run_subworkflow_simple(self):
history_id = self.dataset_populator.new_history()
self._run_jobs(SIMPLE_NESTED_WORKFLOW_YAML, history_id=history_id)
workflow_run_description = """%s
test_data:
outer_input:
value: 1.bed
type: File
""" % SIMPLE_NESTED_WORKFLOW_YAML
self._run_jobs(workflow_run_description, history_id=history_id)
content = self.dataset_populator.get_history_dataset_content(history_id)
self.assertEqual("chr5\t131424298\t131424460\tCCDS4149.1_cds_0_0_chr5_131424299_f\t0\t+\nchr5\t131424298\t131424460\tCCDS4149.1_cds_0_0_chr5_131424299_f\t0\t+\n", content)
@@ -963,6 +970,390 @@ test_data:
time.sleep(5)
self.dataset_populator.wait_for_history(history_id, assert_ok=True)
def test_workflow_output_dataset(self):
history_id = self.dataset_populator.new_history()
summary = self._run_jobs("""
class: GalaxyWorkflow
inputs:
- id: input1
outputs:
- id: wf_output_1
source: first_cat#out_file1
steps:
- tool_id: cat1
label: first_cat
state:
input1:
$link: input1
test_data:
input1: "hello world"
""", history_id=history_id)
workflow_id = summary.workflow_id
invocation_id = summary.invocation_id
invocation_response = self._get("workflows/%s/invocations/%s" % (workflow_id, invocation_id))
self._assert_status_code_is(invocation_response, 200)
invocation = invocation_response.json()
self._assert_has_keys(invocation , "id", "outputs", "output_collections")
assert len(invocation["output_collections"]) == 0
assert len(invocation["outputs"]) == 1
output_content = self.dataset_populator.get_history_dataset_content(history_id, dataset_id=invocation["outputs"]["wf_output_1"]["id"])
assert "hello world" == output_content.strip()
def test_workflow_output_dataset_collection(self):
history_id = self.dataset_populator.new_history()
summary = self._run_jobs("""
class: GalaxyWorkflow
inputs:
- id: input1
type: data_collection_input
collection_type: list
outputs:
- id: wf_output_1
source: first_cat#out_file1
steps:
- tool_id: cat
label: first_cat
state:
input1:
$link: input1
test_data:
input1:
type: list
name: the_dataset_list
elements:
- identifier: el1
value: 1.fastq
type: File
""", history_id=history_id)
workflow_id = summary.workflow_id
invocation_id = summary.invocation_id
invocation_response = self._get("workflows/%s/invocations/%s" % (workflow_id, invocation_id))
self._assert_status_code_is(invocation_response, 200)
invocation = invocation_response.json()
self._assert_has_keys(invocation , "id", "outputs", "output_collections")
assert len(invocation["output_collections"]) == 1
assert len(invocation["outputs"]) == 0
output_content = self.dataset_populator.get_history_collection_details(history_id, content_id=invocation["output_collections"]["wf_output_1"]["id"])
self._assert_has_keys(output_content , "id", "elements")
assert output_content["collection_type"] == "list"
elements = output_content["elements"]
assert len(elements) == 1
elements0 = elements[0]
assert elements0["element_identifier"] == "el1"
def test_worklfow_input_mapping(self):
history_id = self.dataset_populator.new_history()
summary = self._run_jobs("""
class: GalaxyWorkflow
inputs:
- id: input1
outputs:
- id: wf_output_1
source: first_cat#out_file1
steps:
- tool_id: cat
label: first_cat
state:
input1:
$link: input1
test_data:
input1:
type: list
name: the_dataset_list
elements:
- identifier: el1
value: 1.fastq
type: File
- identifier: el2
value: 1.fastq
type: File
""", history_id=history_id)
workflow_id = summary.workflow_id
invocation_id = summary.invocation_id
invocation_response = self._get("workflows/%s/invocations/%s" % (workflow_id, invocation_id))
self._assert_status_code_is(invocation_response, 200)
invocation = invocation_response.json()
self._assert_has_keys(invocation , "id", "outputs", "output_collections")
assert len(invocation["output_collections"]) == 1
assert len(invocation["outputs"]) == 0
output_content = self.dataset_populator.get_history_collection_details(history_id, content_id=invocation["output_collections"]["wf_output_1"]["id"])
self._assert_has_keys(output_content , "id", "elements")
elements = output_content["elements"]
assert len(elements) == 2
elements0 = elements[0]
assert elements0["element_identifier"] == "el1"
@skip_without_tool("collection_creates_pair")
def test_workflow_run_input_mapping_with_output_collections(self):
history_id = self.dataset_populator.new_history()
summary = self._run_jobs("""
class: GalaxyWorkflow
outputs:
- id: wf_output_1
source: split_up#paired_output
steps:
- label: text_input
type: input
- label: split_up
tool_id: collection_creates_pair
state:
input1:
$link: text_input
test_data:
text_input:
type: list
name: the_dataset_list
elements:
- identifier: el1
value: 1.fastq
type: File
- identifier: el2
value: 1.fastq
type: File
""", history_id=history_id)
workflow_id = summary.workflow_id
invocation_id = summary.invocation_id
invocation_response = self._get("workflows/%s/invocations/%s" % (workflow_id, invocation_id))
self._assert_status_code_is(invocation_response, 200)
invocation = invocation_response.json()
self._assert_has_keys(invocation , "id", "outputs", "output_collections")
assert len(invocation["output_collections"]) == 1
assert len(invocation["outputs"]) == 0
output_content = self.dataset_populator.get_history_collection_details(history_id, content_id=invocation["output_collections"]["wf_output_1"]["id"])
self._assert_has_keys(output_content , "id", "elements")
assert output_content["collection_type"] == "list:paired", output_content
elements = output_content["elements"]
assert len(elements) == 2
elements0 = elements[0]
assert elements0["element_identifier"] == "el1"
def test_workflow_run_input_mapping_with_subworkflows(self):
with self.dataset_populator.test_history() as history_id:
summary = self._run_jobs("""%s
test_data:
outer_input:
type: list
name: the_dataset_list
elements:
- identifier: el1
value: 1.fastq
type: File
- identifier: el2
value: 1.fastq
type: File
""" % SIMPLE_NESTED_WORKFLOW_YAML, history_id=history_id)
workflow_id = summary.workflow_id
invocation_id = summary.invocation_id
invocation_response = self._get("workflows/%s/invocations/%s" % (workflow_id, invocation_id))
self._assert_status_code_is(invocation_response, 200)
invocation_response = self._get("workflows/%s/invocations/%s" % (workflow_id, invocation_id))
self._assert_status_code_is(invocation_response, 200)
invocation = invocation_response.json()
self._assert_has_keys(invocation , "id", "outputs", "output_collections")
assert len(invocation["output_collections"]) == 1, invocation
assert len(invocation["outputs"]) == 0
output_content = self.dataset_populator.get_history_collection_details(history_id, content_id=invocation["output_collections"]["outer_output"]["id"])
self._assert_has_keys(output_content , "id", "elements")
assert output_content["collection_type"] == "list", output_content
elements = output_content["elements"]
assert len(elements) == 2
elements0 = elements[0]
assert elements0["element_identifier"] == "el1"
@skip_without_tool("cat_list")
@skip_without_tool("random_lines1")
@skip_without_tool("split")
def test_subworkflow_recover_mapping(self):
with self.dataset_populator.test_history() as history_id:
self._run_jobs("""
class: GalaxyWorkflow
inputs:
- id: outer_input
outputs:
- id: outer_output
source: second_cat#out_file1
steps:
- tool_id: cat1
label: first_cat
state:
input1:
$link: outer_input
- run:
class: GalaxyWorkflow
inputs:
- id: inner_input
outputs:
- id: workflow_output
source: random_lines#out_file1
steps:
- tool_id: random_lines1
label: random_lines
state:
num_lines: 2
input:
$link: inner_input
seed_source:
seed_source_selector: set_seed
seed: asdf
label: nested_workflow
connect:
inner_input: first_cat#out_file1
- tool_id: split
label: split
state:
input1:
$link: nested_workflow#workflow_output
- tool_id: cat_list
label: second_cat
state:
input1:
$link: split#output
test_data:
outer_input:
value: 1.bed
type: File
""", history_id=history_id, wait=True)
self.assertEqual("chr16\t142908\t143003\tCCDS10397.1_cds_0_0_chr16_142909_f\t0\t+\nchr5\t131424298\t131424460\tCCDS4149.1_cds_0_0_chr5_131424299_f\t0\t+\n", self.dataset_populator.get_history_dataset_content(history_id))
@skip_without_tool("cat_list")
@skip_without_tool("random_lines1")
@skip_without_tool("split")
def test_recover_mapping_in_subworkflow(self):
with self.dataset_populator.test_history() as history_id:
self._run_jobs("""
class: GalaxyWorkflow
inputs:
- id: outer_input
outputs:
- id: outer_output
source: second_cat#out_file1
steps:
- tool_id: cat1
label: first_cat
state:
input1:
$link: outer_input
- run:
class: GalaxyWorkflow
inputs:
- id: inner_input
outputs:
- id: workflow_output
source: split#output
steps:
- tool_id: random_lines1
label: random_lines
state:
num_lines: 2
input:
$link: inner_input
seed_source:
seed_source_selector: set_seed
seed: asdf
- tool_id: split
label: split
state:
input1:
$link: random_lines#out_file1
label: nested_workflow
connect:
inner_input: first_cat#out_file1
- tool_id: cat_list
label: second_cat
state:
input1:
$link: nested_workflow#workflow_output
test_data:
outer_input:
value: 1.bed
type: File
""", history_id=history_id, wait=True)
self.assertEqual("chr16\t142908\t143003\tCCDS10397.1_cds_0_0_chr16_142909_f\t0\t+\nchr5\t131424298\t131424460\tCCDS4149.1_cds_0_0_chr5_131424299_f\t0\t+\n", self.dataset_populator.get_history_dataset_content(history_id))
@skip_without_tool("empty_list")
@skip_without_tool("count_list")
@skip_without_tool("random_lines1")
def test_empty_list_mapping(self):
with self.dataset_populator.test_history() as history_id:
self._run_jobs("""
class: GalaxyWorkflow
inputs:
- id: input1
outputs:
- id: count_list
source: count_list#out_file1
steps:
- tool_id: empty_list
label: empty_list
state:
input1:
$link: input1
- tool_id: random_lines1
label: random_lines
state:
num_lines: 2
input:
$link: empty_list#output
seed_source:
seed_source_selector: set_seed
seed: asdf
- tool_id: count_list
label: count_list
state:
input1:
$link: random_lines#out_file1
test_data:
input1:
value: 1.bed
type: File
""", history_id=history_id, wait=True)
self.assertEqual("0\n", self.dataset_populator.get_history_dataset_content(history_id))
@skip_without_tool("empty_list")
@skip_without_tool("count_multi_file")
@skip_without_tool("random_lines1")
def test_empty_list_reduction(self):
with self.dataset_populator.test_history() as history_id:
self._run_jobs("""
class: GalaxyWorkflow
inputs:
- id: input1
outputs:
- id: count_multi_file
source: count_multi_file#out_file1
steps:
- tool_id: empty_list
label: empty_list
state:
input1:
$link: input1
- tool_id: random_lines1
label: random_lines
state:
num_lines: 2
input:
$link: empty_list#output
seed_source:
seed_source_selector: set_seed
seed: asdf
- tool_id: count_multi_file
label: count_multi_file
state:
input1:
$link: random_lines#out_file1
test_data:
input1:
value: 1.bed
type: File
""", history_id=history_id, wait=True)
self.assertEqual("0\n", self.dataset_populator.get_history_dataset_content(history_id))
@skip_without_tool("cat")
def test_cancel_new_workflow_when_history_deleted(self):
with self.dataset_populator.test_history() as history_id:
@@ -1232,7 +1623,7 @@ test_data:
def wait_for_invocation_and_jobs(self, history_id, workflow_id, invocation_id, assert_ok=True):
self.workflow_populator.wait_for_invocation(workflow_id, invocation_id)
time.sleep(.5)
self.dataset_populator.wait_for_history(history_id, assert_ok=assert_ok)
self.dataset_populator.wait_for_history_jobs(history_id, assert_ok=assert_ok)
time.sleep(.5)
def test_cannot_run_inaccessible_workflow(self):
@@ -1351,7 +1742,7 @@ test_data:
value: 1.fastq
type: File
""", history_id=history_id)
content = self.dataset_populator.get_history_dataset_details(history_id, hid=3, wait=True, assert_ok=True)
content = self.dataset_populator.get_history_dataset_details(history_id, hid=4, wait=True, assert_ok=True)
name = content["name"]
assert name == "my new name", name
@@ -2040,19 +2431,8 @@ steps:
self._assert_status_code_is(hda_info_response, 200)
self.assertEqual(hda_info_response.json()["metadata_data_lines"], lines)
def __invoke_workflow(self, history_id, workflow_id, inputs={}, request={}, assert_ok=True):
request["history"] = "hist_id=%s" % history_id,
if inputs:
request["inputs"] = dumps(inputs)
request["inputs_by"] = 'step_index'
url = "workflows/%s/usage" % (workflow_id)
invocation_response = self._post(url, data=request)
if assert_ok:
self._assert_status_code_is(invocation_response, 200)
invocation_id = invocation_response.json()["id"]
return invocation_id
else:
return invocation_response
def __invoke_workflow(self, *args, **kwds):
return self.workflow_populator.invoke_workflow(*args, **kwds)
def __import_workflow(self, workflow_id, deprecated_route=False):
if deprecated_route:
+62 -2
View File
@@ -92,6 +92,18 @@ def skip_without_datatype(extension):
return method_wrapper
def summarize_instance_history_on_error(method):
@wraps(method)
def wrapped_method(api_test_case, *args, **kwds):
try:
method(api_test_case, *args, **kwds)
except Exception:
api_test_case.dataset_populator._summarize_history(api_test_case.history_id)
raise
return wrapped_method
def _raise_skip_if(check):
if check:
from nose.plugins.skip import SkipTest
@@ -151,6 +163,23 @@ class BaseDatasetPopulator(object):
self._summarize_history(history_id)
raise
def wait_for_history_jobs(self, history_id, assert_ok=False, timeout=DEFAULT_TIMEOUT):
query_params = {"history_id": history_id}
def has_active_jobs():
jobs_response = self._get("jobs", query_params)
assert jobs_response.status_code == 200
active_jobs = [j for j in jobs_response.json() if j["state"] in ["new", "upload", "waiting", "queued", "running"]]
if len(active_jobs) == 0:
return True
else:
return None
wait_on(has_active_jobs, "active jobs", timeout=timeout)
if assert_ok:
return self.wait_for_history(history_id, assert_ok=True, timeout=timeout)
def wait_for_job(self, job_id, assert_ok=False, timeout=DEFAULT_TIMEOUT):
return wait_on_state(lambda: self.get_job_details(job_id), assert_ok=assert_ok, timeout=timeout)
@@ -262,6 +291,21 @@ class BaseDatasetPopulator(object):
assert details_response.status_code == 200, details_response.content
return details_response.json()
def run_collection_creates_list(self, history_id, hdca_id):
inputs = {
"input1": {"src": "hdca", "id": hdca_id},
}
self.wait_for_history(history_id, assert_ok=True)
return self.run_tool("collection_creates_list", inputs, history_id)
def run_exit_code_from_file(self, history_id, hdca_id):
exit_code_inputs = {
"input": {'batch': True, 'values': [{"src": "hdca", "id": hdca_id}]},
}
response = self.run_tool("exit_code_from_file", exit_code_inputs, history_id, assert_ok=False).json()
self.wait_for_history(history_id, assert_ok=False)
return response
def __history_content_id(self, history_id, wait=True, **kwds):
if wait:
assert_ok = kwds.get("assert_ok", True)
@@ -270,6 +314,8 @@ class BaseDatasetPopulator(object):
# the last dataset in the history will be fetched.
if "dataset_id" in kwds:
history_content_id = kwds["dataset_id"]
elif "content_id" in kwds:
history_content_id = kwds["content_id"]
elif "dataset" in kwds:
history_content_id = kwds["dataset"]["id"]
else:
@@ -371,7 +417,21 @@ class BaseWorkflowPopulator(object):
""" Wait for a workflow invocation to completely schedule and then history
to be complete. """
self.wait_for_invocation(workflow_id, invocation_id, timeout=timeout)
self.dataset_populator.wait_for_history(history_id, assert_ok=assert_ok, timeout=timeout)
self.dataset_populator.wait_for_history_jobs(history_id, assert_ok=assert_ok, timeout=timeout)
def invoke_workflow(self, history_id, workflow_id, inputs={}, request={}, assert_ok=True):
request["history"] = "hist_id=%s" % history_id,
if inputs:
request["inputs"] = json.dumps(inputs)
request["inputs_by"] = 'step_index'
url = "workflows/%s/usage" % (workflow_id)
invocation_response = self._post(url, data=request)
if assert_ok:
api_asserts.assert_status_code_is(invocation_response, 200)
invocation_id = invocation_response.json()["id"]
return invocation_id
else:
return invocation_response
class WorkflowPopulator(BaseWorkflowPopulator, ImporterGalaxyInterface):
@@ -587,7 +647,7 @@ class BaseDatasetCollectionPopulator(object):
return element_identifiers
def list_identifiers(self, history_id, contents=None):
count = 3 if not contents else len(contents)
count = 3 if contents is None else len(contents)
# Contents can be a list of strings (with name auto-assigned here) or a list of
# 2-tuples of form (name, dataset_content).
if contents and isinstance(contents[0], tuple):
@@ -1,14 +1,16 @@
<tool id="collection_creates_dynamic_nested" name="collection_creates_dynamic_nested" version="0.1.0">
<command>
<command><![CDATA[
echo "A" > oe1_ie1.fq ;
echo "B" > oe1_ie2.fq ;
echo "C" > oe2_ie1.fq ;
echo "D" > oe2_ie2.fq ;
echo "E" > oe3_ie1.fq ;
echo "F" > oe3_ie2.fq
</command>
echo "F" > oe3_ie2.fq ;
sleep '$sleep_time';
]]></command>
<inputs>
<param name="foo" type="text" label="Dummy Parameter" />
<param name="sleep_time" type="integer" label="Sleep Time" value="0" />
</inputs>
<outputs>
<collection name="list_output" type="list:list" label="Duplicate List">
@@ -12,7 +12,7 @@
<param name="foo" type="text" label="Dummy Parameter" />
</inputs>
<outputs>
<collection name="list_output" type="list:list" label="Duplicate List">
<collection name="list_output" type="list:list" label="Failed List">
<!-- Use named regex group to grab pattern
<identifier_0>_<identifier_1>.fq. Here identifier_0 is the list
identifier of the outer list and identifier_1 is the list identifier
@@ -0,0 +1,16 @@
<tool id="count_list" name="count_list">
<description>count the number of items in a list</description>
<command><![CDATA[
echo '${len($input1.keys())}' > '$out_file1'
]]></command>
<inputs>
<param name="input1" type="data_collection" label="Concatenate Dataset" collection_type="list" />
</inputs>
<outputs>
<data name="out_file1" format="txt" />
</outputs>
<tests>
</tests>
<help>
</help>
</tool>
@@ -0,0 +1,16 @@
<tool id="count_multi_file" name="count_multi_file">
<description>count the number of datasets in a multiple file input</description>
<command><![CDATA[
echo '${len($input1)}' > '$out_file1'
]]></command>
<inputs>
<param name="input1" type="data" label="Concatenate Dataset" multiple="true" />
</inputs>
<outputs>
<data name="out_file1" format="txt" />
</outputs>
<tests>
</tests>
<help>
</help>
</tool>
@@ -0,0 +1,22 @@
<tool id="empty_list" name="empty_list" version="0.1.0">
<description>always produce an empty list</description>
<command detect_errors="exit_code">
mkdir outputs;
cd outputs;
</command>
<inputs>
<param name="input1" type="data" format="txt" label="Input Text" />
</inputs>
<outputs>
<collection name="output" type="list" label="lines">
<discover_datasets pattern="__name__" directory="outputs" />
</collection>
</outputs>
<tests>
<test>
<param name="input1" value="simple_lines_both.txt" />
<output_collection name="output" type="list" count="0">
</output_collection>
</test>
</tests>
</tool>
@@ -0,0 +1,29 @@
<tool id="output_filter_with_input" name="output_filter_with_input" version="1.0.0">
<!-- output_filter.xml but with an input -->
<command>
echo "test" > 1;
echo "test" > 2;
echo "test" > 3;
echo "test" > 4;
echo "test" > 5;
</command>
<inputs>
<param name="input_1" type="data" />
<param name="produce_out_1" type="boolean" truevalue="true" falsevalue="false" checked="False" label="Do Filter 1" />
<param name="filter_text_1" type="text" value="1" />
</inputs>
<outputs>
<data format="txt" from_work_dir="1" name="out_1">
<filter>produce_out_1 is True</filter>
</data>
<data format="txt" from_work_dir="2" name="out_2">
<filter>filter_text_1 in ["foo", "bar"]</filter>
<!-- Must pass all filters... -->
<filter>filter_text_1 == "foo"</filter>
</data>
<data format="txt" from_work_dir="3" name="out_3">
</data>
</outputs>
<tests>
</tests>
</tool>
@@ -60,6 +60,7 @@
<tool file="output_format_deprecated_when.xml" />
<tool file="output_format_collection.xml" />
<tool file="output_filter.xml" />
<tool file="output_filter_with_input.xml" />
<tool file="output_filter_exception_1.xml" />
<tool file="output_collection_filter.xml" />
<tool file="output_auto_format.xml" />
@@ -139,6 +140,9 @@
<tool file="for_workflows/mapper.xml" />
<tool file="for_workflows/mapper2.xml" />
<tool file="for_workflows/split.xml" />
<tool file="for_workflows/empty_list.xml" />
<tool file="for_workflows/count_list.xml" />
<tool file="for_workflows/count_multi_file.xml" />
<tool file="for_workflows/create_input_collection.xml" />
<section id="filter" name="For Tours">
+3
View File
@@ -843,6 +843,9 @@ class NavigatesGalaxy(HasDriver):
menu_selection_element = self.wait_for_sizzle_selector_clickable(menu_item_sizzle_selector)
menu_selection_element.click()
def history_panel_click_copy_elements(self):
self.click_history_option("Copy Datasets")
@retry_during_transitions
def histories_click_advanced_search(self):
search_selector = '#standard-search .advanced-search-toggle'
+21
View File
@@ -126,12 +126,33 @@ history_panel:
options_button: '#history-options-button'
options_button_icon: '#history-options-button span.fa-cog'
options_menu: '#history-options-button-menu'
multi_view_button: '#history-view-multi-button'
text:
tooltip_name: 'Click to rename history'
new_name: 'Unnamed history'
new_size: '(empty)'
multi_history_view:
selectors:
_: '.multi-panel-history'
current_label: '.current-label'
create_new_button: '.create-new'
drag_drop_help: '.history-drop-target-help'
history_copy_elements:
selectors:
# Following two don't really work as CSS would only work as jQuery/sizzle I think
# since the page is dynamically generated.
# https://stackoverflow.com/questions/10645552/is-it-possible-to-use-an-input-value-attribute-as-a-css-selector
dataset_checkbox: "input[id='dataset|${id}']"
collection_checkbox: 'input[id="dataset_collection|${id}"]'
new_history_name: '#new_history_name'
copy_button: "input[type='submit']"
done_link: '.donemessage a'
collection_builders:
selectors:
@@ -1,48 +0,0 @@
"""Integration tests for maximum workflow invocation duration configuration option."""
import time
from json import dumps
from base import integration_util
from base.populators import (
DatasetPopulator,
WorkflowPopulator,
)
class MaximumWorkflowInvocationDurationTestCase(integration_util.IntegrationTestCase):
"""Start a Pulsar job."""
framework_tool_and_types = True
def setUp(self):
super(MaximumWorkflowInvocationDurationTestCase, self).setUp()
self.dataset_populator = DatasetPopulator(self.galaxy_interactor)
self.workflow_populator = WorkflowPopulator(self.galaxy_interactor)
@classmethod
def handle_galaxy_config_kwds(cls, config):
config["maximum_workflow_invocation_duration"] = 20
def do_test(self):
workflow = self.workflow_populator.load_workflow_from_resource("test_workflow_pause")
workflow_id = self.workflow_populator.create_workflow(workflow)
history_id = self.dataset_populator.new_history()
hda1 = self.dataset_populator.new_dataset(history_id, content="1 2 3")
index_map = {
'0': dict(src="hda", id=hda1["id"])
}
request = {}
request["history"] = "hist_id=%s" % history_id
request["inputs"] = dumps(index_map)
request["inputs_by"] = 'step_index'
url = "workflows/%s/invocations" % (workflow_id)
invocation_response = self._post(url, data=request)
invocation_url = url + "/" + invocation_response.json()["id"]
time.sleep(5)
state = self._get(invocation_url).json()["state"]
assert state != "failed", state
time.sleep(35)
state = self._get(invocation_url).json()["state"]
assert state == "failed", state
@@ -0,0 +1,92 @@
"""Integration tests for workflow scheduling configuration option."""
import time
from json import dumps
from base import integration_util
from base.populators import (
DatasetCollectionPopulator,
DatasetPopulator,
WorkflowPopulator,
)
class MaximumWorkflowInvocationDurationTestCase(integration_util.IntegrationTestCase):
framework_tool_and_types = True
def setUp(self):
super(MaximumWorkflowInvocationDurationTestCase, self).setUp()
self.dataset_populator = DatasetPopulator(self.galaxy_interactor)
self.workflow_populator = WorkflowPopulator(self.galaxy_interactor)
@classmethod
def handle_galaxy_config_kwds(cls, config):
config["maximum_workflow_invocation_duration"] = 20
def do_test(self):
workflow = self.workflow_populator.load_workflow_from_resource("test_workflow_pause")
workflow_id = self.workflow_populator.create_workflow(workflow)
history_id = self.dataset_populator.new_history()
hda1 = self.dataset_populator.new_dataset(history_id, content="1 2 3")
index_map = {
'0': dict(src="hda", id=hda1["id"])
}
request = {}
request["history"] = "hist_id=%s" % history_id
request["inputs"] = dumps(index_map)
request["inputs_by"] = 'step_index'
url = "workflows/%s/invocations" % (workflow_id)
invocation_response = self._post(url, data=request)
invocation_url = url + "/" + invocation_response.json()["id"]
time.sleep(5)
state = self._get(invocation_url).json()["state"]
assert state != "failed", state
time.sleep(35)
state = self._get(invocation_url).json()["state"]
assert state == "failed", state
class MaximumWorkflowJobsPerSchedulingIterationTestCase(integration_util.IntegrationTestCase):
framework_tool_and_types = True
def setUp(self):
super(MaximumWorkflowJobsPerSchedulingIterationTestCase, self).setUp()
self.dataset_populator = DatasetPopulator(self.galaxy_interactor)
self.workflow_populator = WorkflowPopulator(self.galaxy_interactor)
self.dataset_collection_populator = DatasetCollectionPopulator(self.galaxy_interactor)
@classmethod
def handle_galaxy_config_kwds(cls, config):
config["maximum_workflow_jobs_per_scheduling_iteration"] = 1
def do_test(self):
workflow_id = self.workflow_populator.upload_yaml_workflow("""
class: GalaxyWorkflow
steps:
- type: input_collection
- tool_id: collection_creates_pair
state:
input1:
$link: 0
- tool_id: collection_paired_test
state:
f1:
$link: 1#paired_output
- tool_id: cat_list
state:
input1:
$link: 2#out1
""")
with self.dataset_populator.test_history() as history_id:
hdca1 = self.dataset_collection_populator.create_list_in_history(history_id, contents=["a\nb\nc\nd\n", "e\nf\ng\nh\n"]).json()
self.dataset_populator.wait_for_history(history_id, assert_ok=True)
inputs = {
'0': {"src": "hdca", "id": hdca1["id"]},
}
invocation_id = self.workflow_populator.invoke_workflow(history_id, workflow_id, inputs)
self.workflow_populator.wait_for_workflow(history_id, workflow_id, invocation_id)
self.dataset_populator.wait_for_history(history_id, assert_ok=True)
self.assertEqual("a\nc\nb\nd\ne\ng\nf\nh\n", self.dataset_populator.get_history_dataset_content(history_id, hid=0))
+16 -1
View File
@@ -339,6 +339,14 @@ class SeleniumTestCase(FunctionalTestCase, NavigatesGalaxy, UsesApiTestCaseMixin
with self.main_panel():
self.assert_no_error_message()
@property
def dataset_populator(self):
return SeleniumSessionDatasetPopulator(self)
@property
def dataset_collection_populator(self):
return SeleniumSessionDatasetCollectionPopulator(self)
@property
def workflow_populator(self):
return SeleniumSessionWorkflowPopulator(self)
@@ -465,13 +473,20 @@ class SeleniumSessionGetPostMixin:
"""Mixin for adapting Galaxy testing populators helpers to Selenium session backed bioblend."""
def _get(self, route):
return self.selenium_test_case.api_get(route)
full_url = self.selenium_test_case.build_url("api/" + route, for_selenium=False)
response = requests.get(full_url, cookies=self.selenium_test_case.selenium_to_requests_cookies())
return response
def _post(self, route, data={}):
full_url = self.selenium_test_case.build_url("api/" + route, for_selenium=False)
response = requests.post(full_url, data=data, cookies=self.selenium_test_case.selenium_to_requests_cookies())
return response
def _delete(self, route, data={}):
full_url = self.selenium_test_case.build_url("api/" + route, for_selenium=False)
response = requests.delete(full_url, data=data, cookies=self.selenium_test_case.selenium_to_requests_cookies())
return response
def __url(self, route):
return self._gi.url + "/" + route
@@ -0,0 +1,40 @@
from .framework import (
selenium_test,
SeleniumTestCase
)
class HistoryCopyElementsTestCase(SeleniumTestCase):
ensure_registered = True
@selenium_test
def test_copy_hdca(self):
history_id = self.current_history_id()
input_collection = self.dataset_collection_populator.create_list_in_history(history_id, contents=["0", "1", "0", "1"]).json()
input_hid = input_collection["hid"]
failed_response = self.dataset_populator.run_exit_code_from_file(history_id, input_collection["id"])
failed_collection = failed_response["implicit_collections"][0]
failed_hid = failed_collection["hid"]
self.home()
self.history_panel_wait_for_hid_state(input_hid, "ok")
self.history_panel_wait_for_hid_state(failed_hid, "error")
self.history_panel_click_copy_elements()
with self.main_panel():
self.components.history_copy_elements.collection_checkbox(id=input_collection["id"]).wait_for_and_click()
self.components.history_copy_elements.collection_checkbox(id=failed_collection["id"]).wait_for_and_click()
text_element = self.components.history_copy_elements.new_history_name.wait_for_and_click()
text_element.send_keys("newhistfor_copy_hdca")
self.components.history_copy_elements.copy_button.wait_for_and_click()
self.sleep_for(self.wait_types.UX_TRANSITION)
self.components.history_copy_elements.done_link.wait_for_and_click()
# Okay copied first
self.history_panel_wait_for_hid_state(5, "ok")
# Then 4 datasets and then the failed collection (this was six when coming from the original history)
self.history_panel_wait_for_hid_state(10, "error")
@@ -0,0 +1,24 @@
from .framework import (
selenium_test,
SeleniumTestCase
)
class HistoryMultiViewTestCase(SeleniumTestCase):
ensure_registered = True
@selenium_test
def test_create_new_old_slides_next(self):
history_id = self.current_history_id()
input_collection = self.dataset_collection_populator.create_list_in_history(history_id, contents=["0", "1", "0", "1"]).json()
input_hid = input_collection["hid"]
self.home()
hdca_selector = self.history_panel_wait_for_hid_state(input_hid, "ok")
self.components.history_panel.multi_view_button.wait_for_and_click()
self.components.multi_history_view.create_new_button.wait_for_and_click()
self.components.multi_history_view.drag_drop_help.wait_for_visible()
self.wait_for_visible(hdca_selector)
@@ -0,0 +1,132 @@
import time
from base.api_asserts import assert_status_code_is
from base.populators import flakey
from .framework import (
selenium_test,
SeleniumTestCase
)
class HistoryPanelCollectionsTestCase(SeleniumTestCase):
ensure_registered = True
@selenium_test
def test_mapping_collection_states_terminal(self):
history_id = self.current_history_id()
input_collection = self.dataset_collection_populator.create_list_in_history(history_id, contents=["0", "1", "0", "1"]).json()
input_hid = input_collection["hid"]
failed_response = self.dataset_populator.run_exit_code_from_file(history_id, input_collection["id"])
failed_hid = failed_response["implicit_collections"][0]["hid"]
ok_inputs = {
"input1": {'batch': True, 'values': [{"src": "hdca", "id": input_collection["id"]}]},
"sleep_time": 0,
}
ok_response = self.dataset_populator.run_tool(
"cat_data_and_sleep",
ok_inputs,
history_id,
assert_ok=True,
)
ok_hid = ok_response["implicit_collections"][0]["hid"]
# sleep really shouldn't be needed :(
time.sleep(1)
self.home()
self.history_panel_wait_for_hid_state(input_hid, "ok")
self.history_panel_wait_for_hid_state(failed_hid, "error")
self.history_panel_wait_for_hid_state(ok_hid, "ok")
self.screenshot("history_panel_collections_state_mapping_terminal")
@selenium_test
@flakey # Some times a Paste web thread will stall when jobs are running.
def test_mapping_collection_states_running(self):
history_id = self.current_history_id()
input_collection = self.dataset_collection_populator.create_list_in_history(history_id, contents=["0", "1"]).json()
running_inputs = {
"input1": {'batch': True, 'values': [{"src": "hdca", "id": input_collection["id"]}]},
"sleep_time": 60,
}
running_response = self.dataset_populator.run_tool(
"cat_data_and_sleep",
running_inputs,
history_id,
assert_ok=False,
)
try:
assert_status_code_is(running_response, 200)
running_hid = running_response.json()["implicit_collections"][0]["hid"]
# sleep really shouldn't be needed :(
time.sleep(1)
self.home()
self.history_panel_wait_for_hid_state(running_hid, "running")
self.screenshot("history_panel_collections_state_mapping_running")
finally:
for job in running_response.json()["jobs"]:
self.dataset_populator.cancel_job(job["id"])
@selenium_test
def test_output_collection_states_terminal(self):
history_id = self.current_history_id()
input_collection = self.dataset_collection_populator.create_list_in_history(history_id, contents=["0", "1", "0", "1"]).json()
ok_inputs = {
"input1": {"src": "hdca", "id": input_collection["id"]}
}
ok_response = self.dataset_populator.run_tool(
"collection_creates_list",
ok_inputs,
history_id
)
ok_hid = ok_response["output_collections"][0]["hid"]
failed_response = self.dataset_populator.run_tool(
"collection_creates_dynamic_nested_fail",
{},
history_id,
)
failed_hid = failed_response["output_collections"][0]["hid"]
# sleep really shouldn't be needed :(
time.sleep(1)
self.home()
self.history_panel_wait_for_hid_state(ok_hid, "ok")
self.history_panel_wait_for_hid_state(failed_hid, "error")
self.screenshot("history_panel_collections_state_terminal")
@selenium_test
@flakey # Some times a Paste web thread will stall when jobs are running.
def test_output_collection_states_running(self):
history_id = self.current_history_id()
running_inputs = {
"sleep_time": 180,
}
running_response = self.dataset_populator.run_tool(
"collection_creates_dynamic_nested",
running_inputs,
history_id,
assert_ok=False,
)
try:
assert_status_code_is(running_response, 200)
running_hid = running_response.json()["output_collections"][0]["hid"]
# sleep really shouldn't be needed :(
time.sleep(1)
self.home()
self.history_panel_wait_for_hid_state(running_hid, "running")
self.screenshot("history_panel_collections_state_running")
finally:
for job in running_response.json()["jobs"]:
self.dataset_populator.cancel_job(job["id"])
@@ -117,6 +117,7 @@ class MockCollection(object):
def __init__(self, collection_type, elements):
self.collection_type = collection_type
self.elements = elements
self.populated = True
class MockCollectionElement(object):
+42 -23
View File
@@ -72,13 +72,17 @@ class WorkflowProgressTestCase(unittest.TestCase):
self.invocation, self.inputs_by_step_id, MockModuleInjector(self.progress)
)
def _set_previous_progress(self, outputs_dict):
for step_id, step_value in outputs_dict.items():
def _set_previous_progress(self, outputs):
for i, (step_id, step_value) in enumerate(outputs):
if step_value is not UNSCHEDULED_STEP:
self.progress[step_id] = step_value
workflow_invocation_step = model.WorkflowInvocationStep()
workflow_invocation_step.workflow_step_id = step_id
workflow_invocation_step.state = 'scheduled'
workflow_invocation_step.workflow_step = self._step(i)
self.assertEqual(step_id, self._step(i).id)
# workflow_invocation_step.workflow_invocation = self.invocation
self.invocation.steps.append(workflow_invocation_step)
workflow_invocation_step_state = model.WorkflowRequestStepState()
@@ -89,13 +93,21 @@ class WorkflowProgressTestCase(unittest.TestCase):
def _step(self, index):
return self.invocation.workflow.steps[index]
def _invocation_step(self, index):
if index < len(self.invocation.steps):
return self.invocation.steps[index]
else:
workflow_invocation_step = model.WorkflowInvocationStep()
workflow_invocation_step.workflow_step = self._step(index)
return workflow_invocation_step
def test_connect_data_input(self):
self._setup_workflow(TEST_WORKFLOW_YAML)
hda = model.HistoryDatasetAssociation()
self.inputs_by_step_id = {100: hda}
progress = self._new_workflow_progress()
progress.set_outputs_for_input(self._step(0))
progress.set_outputs_for_input(self._invocation_step(0))
conn = model.WorkflowStepConnection()
conn.output_name = "output"
@@ -108,7 +120,7 @@ class WorkflowProgressTestCase(unittest.TestCase):
self.inputs_by_step_id = {100: hda}
progress = self._new_workflow_progress()
progress.set_outputs_for_input(self._step(0))
progress.set_outputs_for_input(self._invocation_step(0))
replacement = progress.replacement_for_tool_input(self._step(2), MockInput(), "input1")
assert replacement is hda
@@ -118,7 +130,7 @@ class WorkflowProgressTestCase(unittest.TestCase):
hda = model.HistoryDatasetAssociation()
progress = self._new_workflow_progress()
progress.set_step_outputs(self._step(2), {"out1": hda})
progress.set_step_outputs(self._invocation_step(2), {"out1": hda})
conn = model.WorkflowStepConnection()
conn.output_name = "out1"
@@ -128,17 +140,18 @@ class WorkflowProgressTestCase(unittest.TestCase):
def test_remaining_steps_with_progress(self):
self._setup_workflow(TEST_WORKFLOW_YAML)
hda3 = model.HistoryDatasetAssociation()
self._set_previous_progress({
100: {"output": model.HistoryDatasetAssociation()},
101: {"output": model.HistoryDatasetAssociation()},
102: {"out_file1": hda3},
103: {"out_file1": model.HistoryDatasetAssociation()},
104: UNSCHEDULED_STEP,
})
self._set_previous_progress([
(100, {"output": model.HistoryDatasetAssociation()}),
(101, {"output": model.HistoryDatasetAssociation()}),
(102, {"out_file1": hda3}),
(103, {"out_file1": model.HistoryDatasetAssociation()}),
(104, UNSCHEDULED_STEP),
])
progress = self._new_workflow_progress()
steps = progress.remaining_steps()
assert len(steps) == 1
assert steps[0] is self.invocation.workflow.steps[4]
assert len(steps) == 1, steps
step, invocation_step = steps[0]
assert step is self.invocation.workflow.steps[4]
replacement = progress.replacement_for_tool_input(self._step(4), MockInput(), "input1")
assert replacement is hda3
@@ -151,21 +164,27 @@ class WorkflowProgressTestCase(unittest.TestCase):
def test_subworkflow_progress(self):
self._setup_workflow(TEST_SUBWORKFLOW_YAML)
hda = model.HistoryDatasetAssociation()
self._set_previous_progress({
100: {"output": hda},
101: UNSCHEDULED_STEP,
})
self._set_previous_progress([
(100, {"output": hda}),
(101, UNSCHEDULED_STEP),
])
self.invocation.create_subworkflow_invocation_for_step(
self.invocation.workflow.step_by_index(1)
)
progress = self._new_workflow_progress()
remaining_steps = progress.remaining_steps()
subworkflow_step = remaining_steps[0]
(subworkflow_step, subworkflow_invocation_step) = remaining_steps[0]
subworkflow_progress = progress.subworkflow_progress(subworkflow_step)
subworkflow = subworkflow_step.subworkflow
assert subworkflow_progress.workflow_invocation.workflow == subworkflow
subworkflow_input_step = subworkflow.step_by_index(0)
subworkflow_progress.set_outputs_for_input(subworkflow_input_step)
subworkflow_invocation_step = model.WorkflowInvocationStep()
subworkflow_invocation_step.workflow_step_id = subworkflow_input_step.id
subworkflow_invocation_step.state = 'new'
subworkflow_invocation_step.workflow_step = subworkflow_input_step
subworkflow_progress.set_outputs_for_input(subworkflow_invocation_step)
subworkflow_cat_step = subworkflow.step_by_index(1)
@@ -200,7 +219,7 @@ class MockModule(object):
def decode_runtime_state(self, runtime_state):
return True
def recover_mapping(self, step, step_invocations, progress):
step_id = step.id
def recover_mapping(self, invocation_step, progress):
step_id = invocation_step.workflow_step.id
if step_id in self.progress:
progress.set_step_outputs(step, self.progress[step_id])
progress.set_step_outputs(invocation_step, self.progress[step_id])