Implement expression tools and non-data tool outputs.

PR #6925 introduced a GUI for connecting non-data (e.g. integer, boolean, color, etc..) workflow input parameter to tool input parameters (the backend for this was originally added in #1306). That ideally was just the beginning of work toward using such values in structured ways in workflows.

This PR extends tool output handling to allow producing of non-data parameters. These can serve as a source for non-data values in workflows the same work workflow input parameters can. To make such values more easy to produce, this PR also introduces Galaxy expression tools - mirroring functionality regularly used in CWL. These small JavaScript-based tools that consume inputs just like a regular Galaxy tool but that produce dictionary of non-data values.

I think these expressions will be maximally useful when paired with format 2 workflows once we allow users to load arbitrary tools (I make the case more in full here https://github.com/galaxyproject/galaxy/pull/7545#issuecomment-473894424), but I outline some potential uses there as well.

Because there is always a checklist in my PR descriptions:

- Tool definition language and plumbing and datatype for expressing expressions as jobs.
- Allow connecting expression tools to parameters in workflows, will delay evaluation of workflow so calculated value
- Example test expression tools for testing and demonstration.
This commit is contained in:
John Chilton
2019-03-19 17:20:18 -04:00
parent 1d240e4ee2
commit f5c93e868b
23 changed files with 623 additions and 68 deletions
+1
View File
@@ -414,6 +414,7 @@
<!-- End RGenetics Datatypes -->
<datatype extension="ipynb" type="galaxy.datatypes.text:Ipynb" display_in_upload="true"/>
<datatype extension="json" type="galaxy.datatypes.text:Json" display_in_upload="true"/>
<datatype extension="expression.json" type="galaxy.datatypes.text:ExpressionJson" display_in_upload="true"/>
<!-- graph datatypes -->
<datatype extension="xgmml" type="galaxy.datatypes.graph:Xgmml" display_in_upload="true"/>
<datatype extension="sif" type="galaxy.datatypes.graph:Sif" display_in_upload="true"/>
+24
View File
@@ -106,6 +106,30 @@ class Json(Text):
return "JSON file (%s)" % (nice_size(dataset.get_size()))
class ExpressionJson(Json):
""" Represents the non-data input or output to a tool or workflow.
"""
file_ext = "json"
MetadataElement(name="json_type", default=None, desc="JavaScript or JSON type of expression", readonly=True, visible=True, no_value=None)
def set_meta(self, dataset, **kwd):
"""
"""
json_type = "null"
with open(dataset.file_name) as f:
obj = json.load(f)
if isinstance(obj, int):
json_type = "int"
elif isinstance(obj, float):
json_type = "float"
elif isinstance(obj, list):
json_type = "list"
elif isinstance(obj, dict):
json_type = "object"
dataset.metadata.json_type = json_type
@build_sniff_from_prefix
class Ipynb(Json):
file_ext = "ipynb"
+68 -1
View File
@@ -63,6 +63,7 @@ from galaxy.tools.parameters.dataset_matcher import (
from galaxy.tools.parameters.grouping import Conditional, ConditionalWhen, Repeat, Section, UploadDataset
from galaxy.tools.parameters.input_translation import ToolInputTranslator
from galaxy.tools.parameters.meta import expand_meta_parameters
from galaxy.tools.parameters.wrapped_json import json_wrap
from galaxy.tools.parser import (
get_tool_source,
ToolOutputCollectionPart
@@ -2188,6 +2189,72 @@ class OutputParameterJSONTool(Tool):
out.close()
class ExpressionTool(Tool):
requires_js_runtime = True
tool_type = 'expression'
EXPRESSION_INPUTS_NAME = "_expression_inputs_.json"
def parse_command(self, tool_source):
self.command = "cd ../; %s" % expressions.EXPRESSION_SCRIPT_CALL
self.interpreter = None
self._expression = tool_source.parse_expression().strip()
def parse_outputs(self, tool_source):
# Setup self.outputs and self.output_collections
super(ExpressionTool, self).parse_outputs(tool_source)
# Validate these outputs for expression tools.
if len(self.output_collections) != 0:
message = "Expression tools may not declare output collections at this time."
raise Exception(message)
for output in self.outputs.values():
if not hasattr(output, "from_expression"):
message = "Expression tools may not declare output datasets at this time."
raise Exception(message)
def exec_before_job(self, app, inp_data, out_data, param_dict=None):
super(ExpressionTool, self).exec_before_job(app, inp_data, out_data, param_dict=param_dict)
local_working_directory = param_dict["__local_working_directory__"]
expression_inputs_path = os.path.join(local_working_directory, ExpressionTool.EXPRESSION_INPUTS_NAME)
outputs = []
for i, (out_name, data) in enumerate(out_data.iteritems()):
output_def = self.outputs[out_name]
wrapped_data = param_dict.get(out_name)
file_name = str(wrapped_data)
outputs.append(dict(
name=out_name,
from_expression=output_def.from_expression,
path=file_name,
))
if param_dict is None:
raise Exception("Internal error - param_dict is empty.")
job = {}
json_wrap(self.inputs, param_dict, job, handle_files='OBJECT')
expression_inputs = {
'job': job,
'script': self._expression,
'outputs': outputs,
}
expressions.write_evalute_script(os.path.join(local_working_directory))
with open(expression_inputs_path, "w") as f:
json.dump(expression_inputs, f)
def parse_environment_variables(self, tool_source):
""" Setup environment variable for inputs file.
"""
environmnt_variables_raw = super(ExpressionTool, self).parse_environment_variables(tool_source)
expression_script_inputs = dict(
name="GALAXY_EXPRESSION_INPUTS",
template=ExpressionTool.EXPRESSION_INPUTS_NAME,
)
environmnt_variables_raw.append(expression_script_inputs)
return environmnt_variables_raw
class DataSourceTool(OutputParameterJSONTool):
"""
Alternate implementation of Tool for data_source tools -- those that
@@ -2934,7 +3001,7 @@ class FilterFromFileTool(DatabaseOperationTool):
# Populate tool_type to ToolClass mappings
tool_types = {}
for tool_class in [Tool, SetMetadataTool, OutputParameterJSONTool,
for tool_class in [Tool, SetMetadataTool, OutputParameterJSONTool, ExpressionTool,
DataManagerTool, DataSourceTool, AsyncDataSourceTool,
UnzipCollectionTool, ZipCollectionTool, MergeCollectionTool, RelabelFromFileTool, FilterFromFileTool,
BuildListCollectionTool, ExtractDatasetCollectionTool,
+1 -1
View File
@@ -380,7 +380,7 @@ class ToolEvaluator(object):
param_dict['__tool_directory__'] = self.compute_environment.tool_directory()
param_dict['__get_data_table_entry__'] = get_data_table_entry
param_dict['__local_working_directory__'] = self.local_working_directory
# We add access to app here, this allows access to app.config, etc
param_dict['__app__'] = RawObjectWrapper(self.app)
# More convienent access to app.config.new_file_path; we don't need to
+8
View File
@@ -1,12 +1,20 @@
from .evaluation import evaluate
from .sandbox import execjs, interpolate
from .util import jshead, find_engine
from .script import (
write_evalute_script,
EXPRESSION_SCRIPT_CALL,
EXPRESSION_SCRIPT_NAME,
)
__all__ = (
'evaluate',
'execjs',
'EXPRESSION_SCRIPT_CALL',
'EXPRESSION_SCRIPT_NAME',
'find_engine',
'interpolate',
'jshead',
'write_evalute_script',
)
+15
View File
@@ -0,0 +1,15 @@
import os
EXPRESSION_SCRIPT_NAME = "_evaluate_expression_.py"
EXPRESSION_SCRIPT_CALL = "python %s" % EXPRESSION_SCRIPT_NAME
def write_evalute_script(in_directory):
""" Responsible for writing the script that evaluates expressions
in Galaxy jobs.
"""
script = os.path.join(in_directory, EXPRESSION_SCRIPT_NAME)
with open(script, "w") as f:
f.write('from galaxy_ext.expressions.handle_job import run; run()')
return script
+23 -2
View File
@@ -62,8 +62,14 @@ def _json_wrap_input(input, value, handle_files="skip"):
json_value = _data_input_to_path(value)
elif handle_files == "skip":
return SKIP_INPUT
else:
raise NotImplementedError()
elif handle_files == "OBJECT":
if value:
if isinstance(value, list):
value = value[0]
return _hda_to_object(value)
else:
return None
raise NotImplementedError()
elif input_type == "data_collection":
if handle_files == "skip":
return SKIP_INPUT
@@ -88,6 +94,21 @@ def _json_wrap_input(input, value, handle_files="skip"):
return json_value
def _hda_to_object(hda):
hda_dict = hda.to_dict()
metadata_dict = {}
for key, value in hda_dict.items():
if key.startswith("metadata_"):
metadata_dict[key[len("metadata_"):]] = value
return {
'file_ext': hda_dict['file_ext'],
'name': hda_dict['name'],
'metadata': metadata_dict,
}
def _cast_if_not_none(value, cast_to, empty_to_none=False):
# log.debug("value [%s], type[%s]" % (value, type(value)))
if value is None or (empty_to_none and str(value) == ''):
+5
View File
@@ -88,6 +88,11 @@ class ToolSource(object):
""" Return string contianing command to run.
"""
def parse_expression(self):
""" Return string contianing command to run.
"""
return None
@abstractmethod
def parse_environment_variables(self):
""" Return environment variable templates to expose.
+25 -1
View File
@@ -24,12 +24,13 @@ class ToolOutput(ToolOutputBase):
(format, metadata_source, parent)
"""
dict_collection_visible_keys = ['name', 'format', 'label', 'hidden']
dict_collection_visible_keys = ['name', 'format', 'label', 'hidden', 'output_type']
def __init__(self, name, format=None, format_source=None, metadata_source=None,
parent=None, label=None, filters=None, actions=None, hidden=False,
implicit=False):
super(ToolOutput, self).__init__(name, label=label, filters=filters, hidden=hidden)
self.output_type = "data"
self.format = format
self.format_source = format_source
self.metadata_source = metadata_source
@@ -70,6 +71,27 @@ class ToolOutput(ToolOutputBase):
return as_dict
class ToolExpressionOutput(ToolOutputBase):
dict_collection_visible_keys = ('name', 'format', 'label', 'hidden', 'output_type')
def __init__(self, name, output_type, from_expression,
label=None, filters=None, actions=None, hidden=False):
super(ToolExpressionOutput, self).__init__(name, label=label, filters=filters, hidden=hidden)
self.output_type = output_type # JSON type...
self.from_expression = from_expression
self.format = "expression.json" # galaxy.datatypes.text.ExpressionJson.file_ext
self.format_source = None
self.metadata_source = None
self.parent = None
self.actions = actions
# Initialize default values
self.change_format = []
self.implicit = False
self.from_work_dir = None
class ToolOutputCollection(ToolOutputBase):
"""
Represents a HistoryDatasetCollectionAssociation of output datasets produced
@@ -85,6 +107,7 @@ class ToolOutputCollection(ToolOutputBase):
</collection>
<outputs>
"""
dict_collection_visible_keys = ('name', 'format', 'label', 'hidden', 'output_type')
dict_collection_visible_keys = ['name', 'default_format', 'label', 'hidden', 'inherit_format', 'inherit_metadata']
@@ -102,6 +125,7 @@ class ToolOutputCollection(ToolOutputBase):
inherit_metadata=False
):
super(ToolOutputCollection, self).__init__(name, label=label, filters=filters, hidden=hidden)
self.output_type = "collection"
self.collection = True
self.default_format = default_format
self.structure = structure
+48 -1
View File
@@ -22,6 +22,7 @@ from .interface import (
from .output_actions import ToolOutputActionGroup
from .output_collection_def import dataset_collector_descriptions_from_elem
from .output_objects import (
ToolExpressionOutput,
ToolOutput,
ToolOutputCollection,
ToolOutputCollectionStructure
@@ -111,6 +112,12 @@ class XmlToolSource(ToolSource):
command_el = self._command_el
return ((command_el is not None) and command_el.text) or None
def parse_expression(self):
""" Return string contianing command to run.
"""
expression_el = self.root.find("expression")
return ((expression_el is not None) and expression_el.text) or None
def parse_environment_variables(self):
environment_variables_el = self.root.find("environment_variables")
if environment_variables_el is None:
@@ -256,7 +263,12 @@ class XmlToolSource(ToolSource):
for _ in out_elem.findall("data"):
_parse(_)
for collection_elem in out_elem.findall("collection"):
def _parse_expression(output_elem, **kwds):
output_def = self._parse_expression_output(output_elem, tool, **kwds)
data_dict[output_def.name] = output_def
return output_def
def _parse_collection(collection_elem):
name = collection_elem.get("name")
label = xml_text(collection_elem, "label")
default_format = collection_elem.get("format", "data")
@@ -312,6 +324,24 @@ class XmlToolSource(ToolSource):
output_collection.outputs[output_name] = data
output_collections[name] = output_collection
for out_child in out_elem.getchildren():
if out_child.tag == "data":
_parse(out_child)
elif out_child.tag == "collection":
_parse_collection(out_child)
elif out_child.tag == "output":
output_type = out_child.get("type")
if output_type == "data":
_parse(out_child)
elif output_type == "collection":
out_child.attrib["type"] = out_child.get("collection_type")
out_child.attrib["type_source"] = out_child.get("collection_type_source")
_parse_collection(out_child)
else:
_parse_expression(out_child)
else:
log.warn("Unknown output tag encountered [%s]" % out_child.tag)
for output_def in data_dict.values():
outputs[output_def.name] = output_def
return outputs, output_collections
@@ -323,6 +353,7 @@ class XmlToolSource(ToolSource):
default_format="data",
default_format_source=None,
default_metadata_source="",
expression_type=None,
):
output = ToolOutput(data_elem.get("name"))
output_format = data_elem.get("format", default_format)
@@ -347,6 +378,22 @@ class XmlToolSource(ToolSource):
output.dataset_collector_descriptions = dataset_collector_descriptions_from_elem(data_elem, legacy=self.legacy_defaults)
return output
def _parse_expression_output(self, output_elem, tool, **kwds):
output_type = output_elem.get("type")
from_expression = output_elem.get("from")
output = ToolExpressionOutput(
output_elem.get("name"),
output_type,
from_expression,
)
output.path = output_elem.get("value")
output.label = xml_text(output_elem, "label")
output.hidden = string_as_bool(output_elem.get("hidden", ""))
output.actions = ToolOutputActionGroup(output, output_elem.find('actions'))
output.dataset_collector_descriptions = []
return output
def parse_stdio(self):
"""
parse error handling from command and stdio tag
+3
View File
@@ -54,6 +54,9 @@ class YamlToolSource(ToolSource):
def parse_command(self):
return self.root_dict.get("command")
def parse_expression(self):
return self.root_dict.get("expression")
def parse_environment_variables(self):
return []
+134 -60
View File
@@ -67,7 +67,8 @@ the tool menu immediately following the hyperlink for the tool (based on the
</xs:element>
<xs:element name="action" type="ToolAction" minOccurs="0" maxOccurs="1" />
<xs:element name="environment_variables" type="EnvironmentVariables" minOccurs="0" maxOccurs="1"/>
<xs:element name="command" type="Command"/>
<xs:element name="command" type="Command" minOccurs="0" maxOccurs="1"/>
<xs:element name="expression" type="Expression" minOccurs="0" maxOccurs="1"/>
<xs:element name="request_param_translation" type="RequestParameterTranslation" minOccurs="0"/>
<xs:element name="configfiles" type="ConfigFiles" minOccurs="0"/>
<xs:element name="outputs" type="Outputs" minOccurs="0"/>
@@ -2781,6 +2782,33 @@ Prior to Galaxy release 19.01 the stdio block has only been used for non-legacy
</xs:simpleContent>
</xs:complexType>
<xs:complexType name="Expression">
<xs:annotation>
<xs:documentation xml:lang="en"><![CDATA[
For "Expression Tools" (tools with ``tool_type="expression``) this block describes the expression
used to evaluate inputs and produce outputs. The semantics are going to vary based on the value
of "type" specified for this expression block.
]]></xs:documentation>
</xs:annotation>
<xs:simpleContent>
<xs:extension base="xs:string">
<xs:attribute name="type" type="ExpressionType">
<xs:annotation>
<xs:documentation>Type of expression defined by this expression block. The only current valid option is emca5.1 - which will evaluate the expression in a sandbox using node. The option still must be specified to allow a different default in the future.</xs:documentation>
</xs:annotation>
</xs:attribute>
</xs:extension>
</xs:simpleContent>
</xs:complexType>
<xs:simpleType name="ExpressionType">
<xs:annotation>
<xs:documentation xml:lang="en"></xs:documentation>
</xs:annotation>
<xs:restriction base="xs:string">
<xs:enumeration value="emca5.1" />
</xs:restriction>
</xs:simpleType>
<xs:complexType name="ParamOption">
<xs:annotation>
@@ -3684,6 +3712,7 @@ The default is ``galaxy.json``.
</xs:complexType>
<xs:group name="OutputsElement">
<xs:choice>
<xs:element name="output" type="Output" />
<xs:element name="data" type="OutputData"/>
<xs:element name="collection" type="OutputCollection" />
</xs:choice>
@@ -3725,6 +3754,71 @@ pipes or periods (e.g. ``.``).]]></xs:documentation>
</xs:attribute>
</xs:attributeGroup>
<xs:attributeGroup name="OutputDataAttributes">
<xs:attribute name="auto_format" type="PermissiveBoolean">
<xs:annotation>
<xs:documentation xml:lang="en"><![CDATA[
If ``true``, this output will sniffed and its format determined automatically by Galaxy.
]]></xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="format" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">The short name for the output datatype.
The valid values for format can be found in
[/config/datatypes_conf.xml.sample](https://github.com/galaxyproject/galaxy/blob/dev/config/datatypes_conf.xml.sample)
(e.g. ``format="pdf"`` or ``format="fastqsanger"``).</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="default_identifier_source" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">Sets the source of element identifier to the specified input.
This only applies to collections that are mapped over a non-collection input and that have equivalent structures. If this references input elements in conditionals, this value should be qualified (e.g. ``cond|input`` instead of ``input`` if ``input`` is in a conditional with ``name="cond"``).</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="metadata_source" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">This copies the metadata information
from the tool's input dataset. This is particularly useful for interval data
types where the order of the columns is not set.</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="from_work_dir" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">Relative path to a file produced by the
tool in its working directory. Output's contents are set to this file's
contents.</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="hidden" type="xs:boolean" default="false">
<xs:annotation>
<xs:documentation xml:lang="en">Boolean indicating whether to hide
dataset in the history view. (Default is ``false``.)</xs:documentation>
</xs:annotation>
</xs:attribute>
</xs:attributeGroup>
<xs:attributeGroup name="OutputCollectionAttributes">
<xs:attribute name="structured_like" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">This is the name of input collection or
dataset to derive "structure" of the output from (output element count and
identifiers). For instance, if the referenced input has three ordered items with
identifiers ``sample1``, ``sample2``, and ``sample3``. If this references input
elements in conditionals, this value should be qualified (e.g. ``cond|input`` instead
of ``input`` if ``input`` is in a conditional with ``name="cond"``).</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="inherit_format" type="xs:boolean">
<xs:annotation>
<xs:documentation xml:lang="en">If ``structured_like`` is set, inherit
format of outputs from format of corresponding input.</xs:documentation>
</xs:annotation>
</xs:attribute>
</xs:attributeGroup>
<xs:complexType name="OutputData">
<xs:annotation>
<xs:documentation xml:lang="en"><![CDATA[
@@ -3815,49 +3909,7 @@ on Human (hg18)``.
</xs:sequence>
<!-- TODO: add a unique constraint for action. -->
<xs:attributeGroup ref="OutputCommon"/>
<xs:attribute name="auto_format" type="PermissiveBoolean">
<xs:annotation>
<xs:documentation xml:lang="en"><![CDATA[
If ``true``, this output will sniffed and its format determined automatically by Galaxy.
]]></xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="format" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">The short name for the output datatype.
The valid values for format can be found in
[/config/datatypes_conf.xml.sample](https://github.com/galaxyproject/galaxy/blob/dev/config/datatypes_conf.xml.sample)
(e.g. ``format="pdf"`` or ``format="fastqsanger"``).</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="default_identifier_source" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">Sets the source of element identifier to the specified input.
This only applies to collections that are mapped over a non-collection input and that have equivalent structures. If this references input elements in conditionals, this value should be qualified (e.g. ``cond|input`` instead of ``input`` if ``input`` is in a conditional with ``name="cond"``).</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="metadata_source" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">This copies the metadata information
from the tool's input dataset. This is particularly useful for interval data
types where the order of the columns is not set.</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="from_work_dir" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">Relative path to a file produced by the
tool in its working directory. Output's contents are set to this file's
contents.</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="hidden" type="xs:boolean" default="false">
<xs:annotation>
<xs:documentation xml:lang="en">Boolean indicating whether to hide
dataset in the history view. (Default is ``false``.)</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attributeGroup ref="OutputDataAttributes"/>
</xs:complexType>
<xs:group name="OutputCollectionElement">
@@ -3887,6 +3939,7 @@ Creating collections in tools is covered in-depth in
<xs:group ref="OutputCollectionElement" minOccurs="0" maxOccurs="unbounded" />
</xs:sequence>
<xs:attributeGroup ref="OutputCommon"/>
<xs:attributeGroup ref="OutputCollectionAttributes"/>
<xs:attribute name="type" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">Collection type for output (e.g. ``paired``, ``list``, or ``list:list``).</xs:documentation>
@@ -3898,24 +3951,44 @@ Creating collections in tools is covered in-depth in
derive collection's type (e.g. ``collection_type``) from.</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="structured_like" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">This is the name of input collection or
dataset to derive "structure" of the output from (output element count and
identifiers). For instance, if the referenced input has three ordered items with
identifiers ``sample1``, ``sample2``, and ``sample3``. If this references input
elements in conditionals, this value should be qualified (e.g. ``cond|input`` instead
of ``input`` if ``input`` is in a conditional with ``name="cond"``).</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="inherit_format" type="xs:boolean">
<xs:annotation>
<xs:documentation xml:lang="en">If ``structured_like`` is set, inherit
format of outputs from format of corresponding input.</xs:documentation>
</xs:annotation>
</xs:attribute>
</xs:complexType>
<xs:complexType name="Output">
<xs:annotation>
<xs:documentation xml:lang="en"><![CDATA[
This tag describes an output to the tool.
]]></xs:documentation>
</xs:annotation>
<xs:sequence>
<xs:group ref="OutputDataElement" minOccurs="0" maxOccurs="unbounded" />
<xs:group ref="OutputCollectionElement" minOccurs="0" maxOccurs="unbounded" />
</xs:sequence>
<xs:attributeGroup ref="OutputCommon"/>
<xs:attributeGroup ref="OutputCollectionAttributes"/>
<xs:attribute name="type" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">Output type. This could be older more established Galaxy types (e.g. data and collection) - in which case the semantics of this largely reflect the corresponding ``data`` and ``collection`` tags. This could also be newer non-data types such as ``integer`` or ``boolean``.</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="from" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">In expression tools, use this to specify a dictionary value to populate this output from. The semantics may change for other expression types in the future.</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="collection_type" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">Collection type for output (e.g. ``paired``, ``list``, or ``list:list``).</xs:documentation>
</xs:annotation>
</xs:attribute>
<xs:attribute name="collection_type_source" type="xs:string">
<xs:annotation>
<xs:documentation xml:lang="en">This is the name of input collection to
derive collection's type (e.g. ``collection_type``) from.</xs:documentation>
</xs:annotation>
</xs:attribute>
</xs:complexType>
<xs:complexType name="OutputFilter">
<xs:annotation>
<xs:documentation xml:lang="en"><![CDATA[
@@ -5264,6 +5337,7 @@ and ``bibtex`` are the only supported options.</xs:documentation>
<xs:restriction base="xs:string">
<xs:enumeration value="data_source"/>
<xs:enumeration value="manage_data"/>
<xs:enumeration value="expression"/>
</xs:restriction>
</xs:simpleType>
<xs:simpleType name="URLmethodType">
+12
View File
@@ -1,3 +1,4 @@
import json
import logging
import uuid
@@ -367,6 +368,17 @@ class WorkflowProgress(object):
delayed_why = "dependent collection [%s] not yet populated with datasets" % replacement.id
raise modules.DelayedWorkflowEvaluation(why=delayed_why)
is_hda = isinstance(replacement, model.HistoryDatasetAssociation)
if not is_data and is_hda:
if replacement.is_ok:
with open(replacement.file_name, 'r') as f:
replacement = json.load(f)
elif replacement.is_pending:
raise modules.DelayedWorkflowEvaluation()
else:
raise modules.CancelWorkflowEvaluation()
return replacement
def get_replacement_workflow_output(self, workflow_output):
+49
View File
@@ -0,0 +1,49 @@
"""
Execute an external process to evaluate expressions for Galaxy jobs.
Galaxy should be importable on sys.path .
"""
import json
import logging
import os
import sys
# insert *this* galaxy before all others on sys.path
sys.path.insert(1, os.path.abspath(os.path.join(os.path.dirname(__file__), os.pardir, os.pardir)))
# ensure supported version
assert sys.version_info[:2] >= (2, 7) and sys.version_info[:2] <= (2, 7), 'Python version must be 2.7, this is: %s' % sys.version
logging.basicConfig()
log = logging.getLogger(__name__)
from galaxy.tools.expressions import evaluate
try:
from cwltool import expression
except ImportError:
expression = None
def run(environment_path=None):
if expression is None:
raise Exception("Python library cwltool must available to evaluate expressions.")
if environment_path is None:
environment_path = os.environ.get("GALAXY_EXPRESSION_INPUTS")
with open(environment_path, "r") as f:
raw_inputs = json.load(f)
outputs = raw_inputs["outputs"]
inputs = raw_inputs.copy()
del inputs["outputs"]
result = evaluate(None, inputs)
for output in outputs:
path = output["path"]
from_expression = "$(" + output["from_expression"] + ")"
output_value = expression.interpolate(from_expression, result)
with open(path, "w") as f:
json.dump(output_value, f)
+42
View File
@@ -2030,6 +2030,48 @@ class ToolsTestCase(api.ApiTestCase):
output_content = self.dataset_populator.get_history_dataset_content(history_id, dataset=output)
self.assertEqual(output_content.strip(), "123\n456\n456\n0ab")
@skip_without_tool("expression_forty_two")
def test_galaxy_expression_tool_simplest(self):
history_id = self.dataset_populator.new_history()
inputs = {
}
run_response = self._run(
"expression_forty_two", history_id, inputs
)
self._assert_status_code_is(run_response, 200)
self.dataset_populator.wait_for_history(history_id, assert_ok=True)
output_content = self.dataset_populator.get_history_dataset_content(history_id)
self.assertEqual(output_content, "42")
@skip_without_tool("expression_parse_int")
def test_galaxy_expression_tool_simple(self):
history_id = self.dataset_populator.new_history()
inputs = {
'input1': '7',
}
run_response = self._run(
"expression_parse_int", history_id, inputs
)
self._assert_status_code_is(run_response, 200)
self.dataset_populator.wait_for_history(history_id, assert_ok=True)
output_content = self.dataset_populator.get_history_dataset_content(history_id)
self.assertEqual(output_content, "7")
@skip_without_tool("expression_log_line_count")
def test_galaxy_expression_metadata(self):
history_id = self.dataset_populator.new_history()
new_dataset1 = self.dataset_populator.new_dataset(history_id, content='1\n2\n3\n4\n5\n6\n7\n8\n9\n10\n11\n12\n13\n14')
inputs = {
'input1': dataset_to_param(new_dataset1),
}
run_response = self._run(
"expression_log_line_count", history_id, inputs
)
self._assert_status_code_is(run_response, 200)
self.dataset_populator.wait_for_history(history_id, assert_ok=True)
output_content = self.dataset_populator.get_history_dataset_content(history_id)
self.assertEqual(output_content, "3")
def __build_group_list(self, history_id):
response = self.dataset_collection_populator.upload_collection(history_id, "list", elements=[
{
+22 -1
View File
@@ -1965,7 +1965,7 @@ text_input:
type: raw
""", history_id=history_id, wait=True, assert_ok=False)
def test_run_with_text_connection(self):
def test_run_with_text_input_connection(self):
with self.dataset_populator.test_history() as history_id:
self._run_jobs("""
class: GalaxyWorkflow
@@ -1996,6 +1996,27 @@ text_input:
content = self.dataset_populator.get_history_dataset_content(history_id)
self.assertEqual("chrX\t152691446\t152691471\tCCDS14735.1_cds_0_0_chrX_152691447_f\t0\t+\n", content)
def test_run_with_numeric_input_connection(self):
history_id = self.dataset_populator.new_history()
self._run_jobs("""
class: GalaxyWorkflow
steps:
- label: forty_two
tool_id: expression_forty_two
state: {}
- label: consume_expression_parameter
tool_id: cheetah_casting
state:
floattest: 3.14
inttest:
$link: forty_two#out1
test_data: {}
""", history_id=history_id)
self.dataset_populator.wait_for_history(history_id, assert_ok=True)
content = self.dataset_populator.get_history_dataset_content(history_id)
self.assertEquals("43\n4.14\n", content)
@skip_without_tool('cat1')
def test_workflow_rerun_with_use_cached_job(self):
workflow = self.workflow_populator.load_workflow(name="test_for_run")
@@ -0,0 +1,15 @@
<tool name="expression_forty_two" id="expression_forty_two"
version="0.1.0" tool_type="expression">
<description>Parse Int</description>
<expression type="emca5.1">
{return {'output':
42};
}
</expression>
<inputs>
</inputs>
<outputs>
<output type="integer" name="out1" from="output" />
</outputs>
<help>Produces the integer 42.</help>
</tool>
@@ -0,0 +1,14 @@
<tool name="expression_log_line_count" id="expression_log_line_count"
version="0.1.0" tool_type="expression">
<description>Log Lines</description>
<expression type="emca5.1">
{return {'output': Math.max(Math.round(Math.log(parseInt($job.input1.metadata.data_lines))), 1)};}
</expression>
<inputs>
<param type="data" label="Input file to count lines of." name="input1" />
</inputs>
<outputs>
<output type="integer" name="out1" from="output" />
</outputs>
<help></help>
</tool>
@@ -0,0 +1,14 @@
<tool name="expression_parse_int" id="expression_parse_int"
version="0.1.0" tool_type="expression">
<description>Parse Int</description>
<expression type="emca5.1">
{return {'output': parseInt($job.input1)};}
</expression>
<inputs>
<param type="text" label="Text to parse." name="input1" />
</inputs>
<outputs>
<output type="integer" name="out1" from="output" />
</outputs>
<help>Parse an integer from text.</help>
</tool>
+3 -1
View File
@@ -147,8 +147,10 @@
<tool file="collection_creates_dynamic_nested_fail.xml" />
<tool file="collection_cat_group_tag.xml" />
<tool file="collection_cat_group_tag_multiple.xml" />
<tool file="expression_forty_two.xml" />
<tool file="expression_parse_int.xml" />
<tool file="expression_log_line_count.xml" />
<tool file="cheetah_casting.xml" />
<tool file="cheetah_problem_unbound_var.xml" />
<tool file="cheetah_problem_unbound_var_input.xml" />
<tool file="cheetah_problem_syntax_error.xml" />
+48
View File
@@ -0,0 +1,48 @@
import json
import os
import tempfile
import shutil
import subprocess
from galaxy.tools import expressions
THIS_DIRECTORY = os.path.abspath(os.path.dirname(__file__))
TEST_DIRECTORY = os.path.join(THIS_DIRECTORY, os.path.pardir, os.path.pardir)
ROOT_DIRECTORY = os.path.join(TEST_DIRECTORY, os.path.pardir)
LIB_DIRECTORY = os.path.join(ROOT_DIRECTORY, "lib")
def test_run_simple():
test_directory = tempfile.mkdtemp()
try:
environment_path = os.path.join(test_directory, "env.json")
environment = {
'job': {'input1': '7'},
'outputs': [
{'name': 'out1', 'from_expression': "output1", 'path': 'moo'}
],
'script': "{return {'output1': parseInt($job.input1)};}",
}
with open(environment_path, "w") as f:
json.dump(environment, f)
expressions.write_evalute_script(
test_directory,
)
new_env = os.environ.copy()
if "PYTHONPATH" in new_env:
new_env['PYTHONPATH'] = "%s:%s" % (LIB_DIRECTORY, new_env["PYTHONPATH"])
else:
new_env['PYTHONPATH'] = "%s" % (LIB_DIRECTORY)
new_env['GALAXY_EXPRESSION_INPUTS'] = environment_path
p = subprocess.Popen(
args=expressions.EXPRESSION_SCRIPT_CALL,
shell=True,
cwd=test_directory,
env=new_env,
)
assert p.wait() == 0
with open(os.path.join(test_directory, 'moo')) as f:
out_content = f.read()
assert out_content == '7', out_content
finally:
shutil.rmtree(test_directory)
+49
View File
@@ -92,6 +92,39 @@ tests:
compare: sim_size
"""
TOOL_EXPRESSION_XML_1 = """
<tool name="parse_int" id="parse_int" version="0.1.0" tool_type="expression">
<description>Parse Int</description>
<expression>
{return {'output': parseInt($job.input1)};}
</expression>
<inputs>
<input type="text" label="Text to parse." name="input1" />
</inputs>
<outputs>
<output type="integer" name="out1" from="output" />
</outputs>
<help>Parse an integer from text.</help>
</tool>
"""
TOOL_EXPRESSION_YAML_1 = """
class: GalaxyExpressionTool
name: "parse_int"
id: parse_int
version: 1.0.2
expression: "{return {'output': parseInt($job.input1)};}"
inputs:
- name: input1
label: Text to parse
type: text
outputs:
out1:
type: integer
from: "#output"
"""
class BaseLoaderTestCase(unittest.TestCase):
@@ -119,6 +152,22 @@ class BaseLoaderTestCase(unittest.TestCase):
return tool_source
class XmlExpressionLoaderTestCase(BaseLoaderTestCase):
source_file_name = "expression.xml"
source_contents = TOOL_EXPRESSION_XML_1
def test_expression(self):
assert self._tool_source.parse_expression().strip() == "{return {'output': parseInt($job.input1)};}"
def test_tool_type(self):
assert self._tool_source.parse_tool_type() == "expression"
class YamlExpressionLoaderTestCase(BaseLoaderTestCase):
source_file_name = "expression.yml"
source_contents = TOOL_EXPRESSION_XML_1
class XmlLoaderTestCase(BaseLoaderTestCase):
source_file_name = "bwa.xml"
source_contents = TOOL_XML_1