From f5c93e868b7be73f7ced5b7bba8b35b7e41728cc Mon Sep 17 00:00:00 2001 From: John Chilton Date: Thu, 19 Apr 2018 16:08:30 -0400 Subject: [PATCH] Implement expression tools and non-data tool outputs. PR #6925 introduced a GUI for connecting non-data (e.g. integer, boolean, color, etc..) workflow input parameter to tool input parameters (the backend for this was originally added in #1306). That ideally was just the beginning of work toward using such values in structured ways in workflows. This PR extends tool output handling to allow producing of non-data parameters. These can serve as a source for non-data values in workflows the same work workflow input parameters can. To make such values more easy to produce, this PR also introduces Galaxy expression tools - mirroring functionality regularly used in CWL. These small JavaScript-based tools that consume inputs just like a regular Galaxy tool but that produce dictionary of non-data values. I think these expressions will be maximally useful when paired with format 2 workflows once we allow users to load arbitrary tools (I make the case more in full here https://github.com/galaxyproject/galaxy/pull/7545#issuecomment-473894424), but I outline some potential uses there as well. Because there is always a checklist in my PR descriptions: - Tool definition language and plumbing and datatype for expressing expressions as jobs. - Allow connecting expression tools to parameters in workflows, will delay evaluation of workflow so calculated value - Example test expression tools for testing and demonstration. --- config/datatypes_conf.xml.sample | 1 + lib/galaxy/datatypes/text.py | 24 +++ lib/galaxy/tools/__init__.py | 69 ++++++- lib/galaxy/tools/evaluation.py | 2 +- lib/galaxy/tools/expressions/__init__.py | 8 + lib/galaxy/tools/expressions/script.py | 15 ++ lib/galaxy/tools/parameters/wrapped_json.py | 25 ++- lib/galaxy/tools/parser/interface.py | 5 + lib/galaxy/tools/parser/output_objects.py | 26 ++- lib/galaxy/tools/parser/xml.py | 49 ++++- lib/galaxy/tools/parser/yaml.py | 3 + lib/galaxy/tools/xsd/galaxy.xsd | 194 ++++++++++++------ lib/galaxy/workflow/run.py | 12 ++ lib/galaxy_ext/expressions/__init__.py | 0 lib/galaxy_ext/expressions/handle_job.py | 49 +++++ test/api/test_tools.py | 42 ++++ test/api/test_workflows.py | 23 ++- .../functional/tools/expression_forty_two.xml | 15 ++ .../tools/expression_log_line_count.xml | 14 ++ .../functional/tools/expression_parse_int.xml | 14 ++ test/functional/tools/samples_tool_conf.xml | 4 +- test/unit/jobs/test_expression_run.py | 48 +++++ test/unit/tools/test_parsing.py | 49 +++++ 23 files changed, 623 insertions(+), 68 deletions(-) create mode 100644 lib/galaxy/tools/expressions/script.py create mode 100644 lib/galaxy_ext/expressions/__init__.py create mode 100644 lib/galaxy_ext/expressions/handle_job.py create mode 100644 test/functional/tools/expression_forty_two.xml create mode 100644 test/functional/tools/expression_log_line_count.xml create mode 100644 test/functional/tools/expression_parse_int.xml create mode 100644 test/unit/jobs/test_expression_run.py diff --git a/config/datatypes_conf.xml.sample b/config/datatypes_conf.xml.sample index 76fdb0c7ee5..12f4fae5d2f 100644 --- a/config/datatypes_conf.xml.sample +++ b/config/datatypes_conf.xml.sample @@ -414,6 +414,7 @@ + diff --git a/lib/galaxy/datatypes/text.py b/lib/galaxy/datatypes/text.py index c2e71444cf1..c563c2d318c 100644 --- a/lib/galaxy/datatypes/text.py +++ b/lib/galaxy/datatypes/text.py @@ -106,6 +106,30 @@ class Json(Text): return "JSON file (%s)" % (nice_size(dataset.get_size())) +class ExpressionJson(Json): + """ Represents the non-data input or output to a tool or workflow. + """ + file_ext = "json" + MetadataElement(name="json_type", default=None, desc="JavaScript or JSON type of expression", readonly=True, visible=True, no_value=None) + + def set_meta(self, dataset, **kwd): + """ + """ + json_type = "null" + with open(dataset.file_name) as f: + obj = json.load(f) + if isinstance(obj, int): + json_type = "int" + elif isinstance(obj, float): + json_type = "float" + elif isinstance(obj, list): + json_type = "list" + elif isinstance(obj, dict): + json_type = "object" + + dataset.metadata.json_type = json_type + + @build_sniff_from_prefix class Ipynb(Json): file_ext = "ipynb" diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index 19618bf4ca2..f63f33a94b1 100755 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -63,6 +63,7 @@ from galaxy.tools.parameters.dataset_matcher import ( from galaxy.tools.parameters.grouping import Conditional, ConditionalWhen, Repeat, Section, UploadDataset from galaxy.tools.parameters.input_translation import ToolInputTranslator from galaxy.tools.parameters.meta import expand_meta_parameters +from galaxy.tools.parameters.wrapped_json import json_wrap from galaxy.tools.parser import ( get_tool_source, ToolOutputCollectionPart @@ -2188,6 +2189,72 @@ class OutputParameterJSONTool(Tool): out.close() +class ExpressionTool(Tool): + requires_js_runtime = True + tool_type = 'expression' + EXPRESSION_INPUTS_NAME = "_expression_inputs_.json" + + def parse_command(self, tool_source): + self.command = "cd ../; %s" % expressions.EXPRESSION_SCRIPT_CALL + self.interpreter = None + self._expression = tool_source.parse_expression().strip() + + def parse_outputs(self, tool_source): + # Setup self.outputs and self.output_collections + super(ExpressionTool, self).parse_outputs(tool_source) + + # Validate these outputs for expression tools. + if len(self.output_collections) != 0: + message = "Expression tools may not declare output collections at this time." + raise Exception(message) + for output in self.outputs.values(): + if not hasattr(output, "from_expression"): + message = "Expression tools may not declare output datasets at this time." + raise Exception(message) + + def exec_before_job(self, app, inp_data, out_data, param_dict=None): + super(ExpressionTool, self).exec_before_job(app, inp_data, out_data, param_dict=param_dict) + local_working_directory = param_dict["__local_working_directory__"] + expression_inputs_path = os.path.join(local_working_directory, ExpressionTool.EXPRESSION_INPUTS_NAME) + + outputs = [] + for i, (out_name, data) in enumerate(out_data.iteritems()): + output_def = self.outputs[out_name] + wrapped_data = param_dict.get(out_name) + file_name = str(wrapped_data) + + outputs.append(dict( + name=out_name, + from_expression=output_def.from_expression, + path=file_name, + )) + + if param_dict is None: + raise Exception("Internal error - param_dict is empty.") + + job = {} + json_wrap(self.inputs, param_dict, job, handle_files='OBJECT') + expression_inputs = { + 'job': job, + 'script': self._expression, + 'outputs': outputs, + } + expressions.write_evalute_script(os.path.join(local_working_directory)) + with open(expression_inputs_path, "w") as f: + json.dump(expression_inputs, f) + + def parse_environment_variables(self, tool_source): + """ Setup environment variable for inputs file. + """ + environmnt_variables_raw = super(ExpressionTool, self).parse_environment_variables(tool_source) + expression_script_inputs = dict( + name="GALAXY_EXPRESSION_INPUTS", + template=ExpressionTool.EXPRESSION_INPUTS_NAME, + ) + environmnt_variables_raw.append(expression_script_inputs) + return environmnt_variables_raw + + class DataSourceTool(OutputParameterJSONTool): """ Alternate implementation of Tool for data_source tools -- those that @@ -2934,7 +3001,7 @@ class FilterFromFileTool(DatabaseOperationTool): # Populate tool_type to ToolClass mappings tool_types = {} -for tool_class in [Tool, SetMetadataTool, OutputParameterJSONTool, +for tool_class in [Tool, SetMetadataTool, OutputParameterJSONTool, ExpressionTool, DataManagerTool, DataSourceTool, AsyncDataSourceTool, UnzipCollectionTool, ZipCollectionTool, MergeCollectionTool, RelabelFromFileTool, FilterFromFileTool, BuildListCollectionTool, ExtractDatasetCollectionTool, diff --git a/lib/galaxy/tools/evaluation.py b/lib/galaxy/tools/evaluation.py index b244364ac6e..9e56df5e56c 100644 --- a/lib/galaxy/tools/evaluation.py +++ b/lib/galaxy/tools/evaluation.py @@ -380,7 +380,7 @@ class ToolEvaluator(object): param_dict['__tool_directory__'] = self.compute_environment.tool_directory() param_dict['__get_data_table_entry__'] = get_data_table_entry - + param_dict['__local_working_directory__'] = self.local_working_directory # We add access to app here, this allows access to app.config, etc param_dict['__app__'] = RawObjectWrapper(self.app) # More convienent access to app.config.new_file_path; we don't need to diff --git a/lib/galaxy/tools/expressions/__init__.py b/lib/galaxy/tools/expressions/__init__.py index b7519e5f57c..b731044825a 100644 --- a/lib/galaxy/tools/expressions/__init__.py +++ b/lib/galaxy/tools/expressions/__init__.py @@ -1,12 +1,20 @@ from .evaluation import evaluate from .sandbox import execjs, interpolate from .util import jshead, find_engine +from .script import ( + write_evalute_script, + EXPRESSION_SCRIPT_CALL, + EXPRESSION_SCRIPT_NAME, +) __all__ = ( 'evaluate', 'execjs', + 'EXPRESSION_SCRIPT_CALL', + 'EXPRESSION_SCRIPT_NAME', 'find_engine', 'interpolate', 'jshead', + 'write_evalute_script', ) diff --git a/lib/galaxy/tools/expressions/script.py b/lib/galaxy/tools/expressions/script.py new file mode 100644 index 00000000000..0cb18ae7464 --- /dev/null +++ b/lib/galaxy/tools/expressions/script.py @@ -0,0 +1,15 @@ +import os + +EXPRESSION_SCRIPT_NAME = "_evaluate_expression_.py" +EXPRESSION_SCRIPT_CALL = "python %s" % EXPRESSION_SCRIPT_NAME + + +def write_evalute_script(in_directory): + """ Responsible for writing the script that evaluates expressions + in Galaxy jobs. + """ + script = os.path.join(in_directory, EXPRESSION_SCRIPT_NAME) + with open(script, "w") as f: + f.write('from galaxy_ext.expressions.handle_job import run; run()') + + return script diff --git a/lib/galaxy/tools/parameters/wrapped_json.py b/lib/galaxy/tools/parameters/wrapped_json.py index c045c9595ef..69549943143 100644 --- a/lib/galaxy/tools/parameters/wrapped_json.py +++ b/lib/galaxy/tools/parameters/wrapped_json.py @@ -62,8 +62,14 @@ def _json_wrap_input(input, value, handle_files="skip"): json_value = _data_input_to_path(value) elif handle_files == "skip": return SKIP_INPUT - else: - raise NotImplementedError() + elif handle_files == "OBJECT": + if value: + if isinstance(value, list): + value = value[0] + return _hda_to_object(value) + else: + return None + raise NotImplementedError() elif input_type == "data_collection": if handle_files == "skip": return SKIP_INPUT @@ -88,6 +94,21 @@ def _json_wrap_input(input, value, handle_files="skip"): return json_value +def _hda_to_object(hda): + hda_dict = hda.to_dict() + metadata_dict = {} + + for key, value in hda_dict.items(): + if key.startswith("metadata_"): + metadata_dict[key[len("metadata_"):]] = value + + return { + 'file_ext': hda_dict['file_ext'], + 'name': hda_dict['name'], + 'metadata': metadata_dict, + } + + def _cast_if_not_none(value, cast_to, empty_to_none=False): # log.debug("value [%s], type[%s]" % (value, type(value))) if value is None or (empty_to_none and str(value) == ''): diff --git a/lib/galaxy/tools/parser/interface.py b/lib/galaxy/tools/parser/interface.py index 2cd31a1aa6f..bdc979dccfc 100644 --- a/lib/galaxy/tools/parser/interface.py +++ b/lib/galaxy/tools/parser/interface.py @@ -88,6 +88,11 @@ class ToolSource(object): """ Return string contianing command to run. """ + def parse_expression(self): + """ Return string contianing command to run. + """ + return None + @abstractmethod def parse_environment_variables(self): """ Return environment variable templates to expose. diff --git a/lib/galaxy/tools/parser/output_objects.py b/lib/galaxy/tools/parser/output_objects.py index bfe51934ff9..805335a8358 100644 --- a/lib/galaxy/tools/parser/output_objects.py +++ b/lib/galaxy/tools/parser/output_objects.py @@ -24,12 +24,13 @@ class ToolOutput(ToolOutputBase): (format, metadata_source, parent) """ - dict_collection_visible_keys = ['name', 'format', 'label', 'hidden'] + dict_collection_visible_keys = ['name', 'format', 'label', 'hidden', 'output_type'] def __init__(self, name, format=None, format_source=None, metadata_source=None, parent=None, label=None, filters=None, actions=None, hidden=False, implicit=False): super(ToolOutput, self).__init__(name, label=label, filters=filters, hidden=hidden) + self.output_type = "data" self.format = format self.format_source = format_source self.metadata_source = metadata_source @@ -70,6 +71,27 @@ class ToolOutput(ToolOutputBase): return as_dict +class ToolExpressionOutput(ToolOutputBase): + dict_collection_visible_keys = ('name', 'format', 'label', 'hidden', 'output_type') + + def __init__(self, name, output_type, from_expression, + label=None, filters=None, actions=None, hidden=False): + super(ToolExpressionOutput, self).__init__(name, label=label, filters=filters, hidden=hidden) + self.output_type = output_type # JSON type... + self.from_expression = from_expression + self.format = "expression.json" # galaxy.datatypes.text.ExpressionJson.file_ext + + self.format_source = None + self.metadata_source = None + self.parent = None + self.actions = actions + + # Initialize default values + self.change_format = [] + self.implicit = False + self.from_work_dir = None + + class ToolOutputCollection(ToolOutputBase): """ Represents a HistoryDatasetCollectionAssociation of output datasets produced @@ -85,6 +107,7 @@ class ToolOutputCollection(ToolOutputBase): """ + dict_collection_visible_keys = ('name', 'format', 'label', 'hidden', 'output_type') dict_collection_visible_keys = ['name', 'default_format', 'label', 'hidden', 'inherit_format', 'inherit_metadata'] @@ -102,6 +125,7 @@ class ToolOutputCollection(ToolOutputBase): inherit_metadata=False ): super(ToolOutputCollection, self).__init__(name, label=label, filters=filters, hidden=hidden) + self.output_type = "collection" self.collection = True self.default_format = default_format self.structure = structure diff --git a/lib/galaxy/tools/parser/xml.py b/lib/galaxy/tools/parser/xml.py index 997eb760406..4d67278c26f 100644 --- a/lib/galaxy/tools/parser/xml.py +++ b/lib/galaxy/tools/parser/xml.py @@ -22,6 +22,7 @@ from .interface import ( from .output_actions import ToolOutputActionGroup from .output_collection_def import dataset_collector_descriptions_from_elem from .output_objects import ( + ToolExpressionOutput, ToolOutput, ToolOutputCollection, ToolOutputCollectionStructure @@ -111,6 +112,12 @@ class XmlToolSource(ToolSource): command_el = self._command_el return ((command_el is not None) and command_el.text) or None + def parse_expression(self): + """ Return string contianing command to run. + """ + expression_el = self.root.find("expression") + return ((expression_el is not None) and expression_el.text) or None + def parse_environment_variables(self): environment_variables_el = self.root.find("environment_variables") if environment_variables_el is None: @@ -256,7 +263,12 @@ class XmlToolSource(ToolSource): for _ in out_elem.findall("data"): _parse(_) - for collection_elem in out_elem.findall("collection"): + def _parse_expression(output_elem, **kwds): + output_def = self._parse_expression_output(output_elem, tool, **kwds) + data_dict[output_def.name] = output_def + return output_def + + def _parse_collection(collection_elem): name = collection_elem.get("name") label = xml_text(collection_elem, "label") default_format = collection_elem.get("format", "data") @@ -312,6 +324,24 @@ class XmlToolSource(ToolSource): output_collection.outputs[output_name] = data output_collections[name] = output_collection + for out_child in out_elem.getchildren(): + if out_child.tag == "data": + _parse(out_child) + elif out_child.tag == "collection": + _parse_collection(out_child) + elif out_child.tag == "output": + output_type = out_child.get("type") + if output_type == "data": + _parse(out_child) + elif output_type == "collection": + out_child.attrib["type"] = out_child.get("collection_type") + out_child.attrib["type_source"] = out_child.get("collection_type_source") + _parse_collection(out_child) + else: + _parse_expression(out_child) + else: + log.warn("Unknown output tag encountered [%s]" % out_child.tag) + for output_def in data_dict.values(): outputs[output_def.name] = output_def return outputs, output_collections @@ -323,6 +353,7 @@ class XmlToolSource(ToolSource): default_format="data", default_format_source=None, default_metadata_source="", + expression_type=None, ): output = ToolOutput(data_elem.get("name")) output_format = data_elem.get("format", default_format) @@ -347,6 +378,22 @@ class XmlToolSource(ToolSource): output.dataset_collector_descriptions = dataset_collector_descriptions_from_elem(data_elem, legacy=self.legacy_defaults) return output + def _parse_expression_output(self, output_elem, tool, **kwds): + output_type = output_elem.get("type") + from_expression = output_elem.get("from") + output = ToolExpressionOutput( + output_elem.get("name"), + output_type, + from_expression, + ) + output.path = output_elem.get("value") + output.label = xml_text(output_elem, "label") + + output.hidden = string_as_bool(output_elem.get("hidden", "")) + output.actions = ToolOutputActionGroup(output, output_elem.find('actions')) + output.dataset_collector_descriptions = [] + return output + def parse_stdio(self): """ parse error handling from command and stdio tag diff --git a/lib/galaxy/tools/parser/yaml.py b/lib/galaxy/tools/parser/yaml.py index 63e642e5648..52ae9d4cb07 100644 --- a/lib/galaxy/tools/parser/yaml.py +++ b/lib/galaxy/tools/parser/yaml.py @@ -54,6 +54,9 @@ class YamlToolSource(ToolSource): def parse_command(self): return self.root_dict.get("command") + def parse_expression(self): + return self.root_dict.get("expression") + def parse_environment_variables(self): return [] diff --git a/lib/galaxy/tools/xsd/galaxy.xsd b/lib/galaxy/tools/xsd/galaxy.xsd index 17e3191df80..a298cc13dd1 100644 --- a/lib/galaxy/tools/xsd/galaxy.xsd +++ b/lib/galaxy/tools/xsd/galaxy.xsd @@ -67,7 +67,8 @@ the tool menu immediately following the hyperlink for the tool (based on the - + + @@ -2781,6 +2782,33 @@ Prior to Galaxy release 19.01 the stdio block has only been used for non-legacy + + + + + + + + + Type of expression defined by this expression block. The only current valid option is emca5.1 - which will evaluate the expression in a sandbox using node. The option still must be specified to allow a different default in the future. + + + + + + + + + + + + + + @@ -3684,6 +3712,7 @@ The default is ``galaxy.json``. + @@ -3725,6 +3754,71 @@ pipes or periods (e.g. ``.``).]]> + + + + + + + + + The short name for the output datatype. +The valid values for format can be found in +[/config/datatypes_conf.xml.sample](https://github.com/galaxyproject/galaxy/blob/dev/config/datatypes_conf.xml.sample) +(e.g. ``format="pdf"`` or ``format="fastqsanger"``). + + + + + Sets the source of element identifier to the specified input. +This only applies to collections that are mapped over a non-collection input and that have equivalent structures. If this references input elements in conditionals, this value should be qualified (e.g. ``cond|input`` instead of ``input`` if ``input`` is in a conditional with ``name="cond"``). + + + + + This copies the metadata information +from the tool's input dataset. This is particularly useful for interval data +types where the order of the columns is not set. + + + + + Relative path to a file produced by the +tool in its working directory. Output's contents are set to this file's +contents. + + + + + Boolean indicating whether to hide +dataset in the history view. (Default is ``false``.) + + + + + + + + This is the name of input collection or +dataset to derive "structure" of the output from (output element count and +identifiers). For instance, if the referenced input has three ordered items with +identifiers ``sample1``, ``sample2``, and ``sample3``. If this references input +elements in conditionals, this value should be qualified (e.g. ``cond|input`` instead +of ``input`` if ``input`` is in a conditional with ``name="cond"``). + + + + + If ``structured_like`` is set, inherit +format of outputs from format of corresponding input. + + + + - - - - - - - - The short name for the output datatype. -The valid values for format can be found in -[/config/datatypes_conf.xml.sample](https://github.com/galaxyproject/galaxy/blob/dev/config/datatypes_conf.xml.sample) -(e.g. ``format="pdf"`` or ``format="fastqsanger"``). - - - - - Sets the source of element identifier to the specified input. -This only applies to collections that are mapped over a non-collection input and that have equivalent structures. If this references input elements in conditionals, this value should be qualified (e.g. ``cond|input`` instead of ``input`` if ``input`` is in a conditional with ``name="cond"``). - - - - - This copies the metadata information -from the tool's input dataset. This is particularly useful for interval data -types where the order of the columns is not set. - - - - - Relative path to a file produced by the -tool in its working directory. Output's contents are set to this file's -contents. - - - - - Boolean indicating whether to hide -dataset in the history view. (Default is ``false``.) - - + @@ -3887,6 +3939,7 @@ Creating collections in tools is covered in-depth in + Collection type for output (e.g. ``paired``, ``list``, or ``list:list``). @@ -3898,24 +3951,44 @@ Creating collections in tools is covered in-depth in derive collection's type (e.g. ``collection_type``) from. - - - This is the name of input collection or -dataset to derive "structure" of the output from (output element count and -identifiers). For instance, if the referenced input has three ordered items with -identifiers ``sample1``, ``sample2``, and ``sample3``. If this references input -elements in conditionals, this value should be qualified (e.g. ``cond|input`` instead -of ``input`` if ``input`` is in a conditional with ``name="cond"``). - - - - - If ``structured_like`` is set, inherit -format of outputs from format of corresponding input. - - + + + + + + + + + + + + + Output type. This could be older more established Galaxy types (e.g. data and collection) - in which case the semantics of this largely reflect the corresponding ``data`` and ``collection`` tags. This could also be newer non-data types such as ``integer`` or ``boolean``. + + + + + In expression tools, use this to specify a dictionary value to populate this output from. The semantics may change for other expression types in the future. + + + + + Collection type for output (e.g. ``paired``, ``list``, or ``list:list``). + + + + + This is the name of input collection to +derive collection's type (e.g. ``collection_type``) from. + + + + + diff --git a/lib/galaxy/workflow/run.py b/lib/galaxy/workflow/run.py index b75bee7ecac..961a673e9f4 100644 --- a/lib/galaxy/workflow/run.py +++ b/lib/galaxy/workflow/run.py @@ -1,3 +1,4 @@ +import json import logging import uuid @@ -367,6 +368,17 @@ class WorkflowProgress(object): delayed_why = "dependent collection [%s] not yet populated with datasets" % replacement.id raise modules.DelayedWorkflowEvaluation(why=delayed_why) + + is_hda = isinstance(replacement, model.HistoryDatasetAssociation) + if not is_data and is_hda: + if replacement.is_ok: + with open(replacement.file_name, 'r') as f: + replacement = json.load(f) + elif replacement.is_pending: + raise modules.DelayedWorkflowEvaluation() + else: + raise modules.CancelWorkflowEvaluation() + return replacement def get_replacement_workflow_output(self, workflow_output): diff --git a/lib/galaxy_ext/expressions/__init__.py b/lib/galaxy_ext/expressions/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/lib/galaxy_ext/expressions/handle_job.py b/lib/galaxy_ext/expressions/handle_job.py new file mode 100644 index 00000000000..4cfc3b52dc4 --- /dev/null +++ b/lib/galaxy_ext/expressions/handle_job.py @@ -0,0 +1,49 @@ +""" +Execute an external process to evaluate expressions for Galaxy jobs. + +Galaxy should be importable on sys.path . +""" + +import json +import logging +import os +import sys + +# insert *this* galaxy before all others on sys.path +sys.path.insert(1, os.path.abspath(os.path.join(os.path.dirname(__file__), os.pardir, os.pardir))) + +# ensure supported version +assert sys.version_info[:2] >= (2, 7) and sys.version_info[:2] <= (2, 7), 'Python version must be 2.7, this is: %s' % sys.version + +logging.basicConfig() +log = logging.getLogger(__name__) + +from galaxy.tools.expressions import evaluate + +try: + from cwltool import expression +except ImportError: + expression = None + + +def run(environment_path=None): + if expression is None: + raise Exception("Python library cwltool must available to evaluate expressions.") + + if environment_path is None: + environment_path = os.environ.get("GALAXY_EXPRESSION_INPUTS") + with open(environment_path, "r") as f: + raw_inputs = json.load(f) + + outputs = raw_inputs["outputs"] + inputs = raw_inputs.copy() + del inputs["outputs"] + + result = evaluate(None, inputs) + + for output in outputs: + path = output["path"] + from_expression = "$(" + output["from_expression"] + ")" + output_value = expression.interpolate(from_expression, result) + with open(path, "w") as f: + json.dump(output_value, f) diff --git a/test/api/test_tools.py b/test/api/test_tools.py index d1c1e275a80..d2900e5a8d0 100644 --- a/test/api/test_tools.py +++ b/test/api/test_tools.py @@ -2030,6 +2030,48 @@ class ToolsTestCase(api.ApiTestCase): output_content = self.dataset_populator.get_history_dataset_content(history_id, dataset=output) self.assertEqual(output_content.strip(), "123\n456\n456\n0ab") + @skip_without_tool("expression_forty_two") + def test_galaxy_expression_tool_simplest(self): + history_id = self.dataset_populator.new_history() + inputs = { + } + run_response = self._run( + "expression_forty_two", history_id, inputs + ) + self._assert_status_code_is(run_response, 200) + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + output_content = self.dataset_populator.get_history_dataset_content(history_id) + self.assertEqual(output_content, "42") + + @skip_without_tool("expression_parse_int") + def test_galaxy_expression_tool_simple(self): + history_id = self.dataset_populator.new_history() + inputs = { + 'input1': '7', + } + run_response = self._run( + "expression_parse_int", history_id, inputs + ) + self._assert_status_code_is(run_response, 200) + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + output_content = self.dataset_populator.get_history_dataset_content(history_id) + self.assertEqual(output_content, "7") + + @skip_without_tool("expression_log_line_count") + def test_galaxy_expression_metadata(self): + history_id = self.dataset_populator.new_history() + new_dataset1 = self.dataset_populator.new_dataset(history_id, content='1\n2\n3\n4\n5\n6\n7\n8\n9\n10\n11\n12\n13\n14') + inputs = { + 'input1': dataset_to_param(new_dataset1), + } + run_response = self._run( + "expression_log_line_count", history_id, inputs + ) + self._assert_status_code_is(run_response, 200) + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + output_content = self.dataset_populator.get_history_dataset_content(history_id) + self.assertEqual(output_content, "3") + def __build_group_list(self, history_id): response = self.dataset_collection_populator.upload_collection(history_id, "list", elements=[ { diff --git a/test/api/test_workflows.py b/test/api/test_workflows.py index 61f2c558be9..5ab72f3341c 100644 --- a/test/api/test_workflows.py +++ b/test/api/test_workflows.py @@ -1965,7 +1965,7 @@ text_input: type: raw """, history_id=history_id, wait=True, assert_ok=False) - def test_run_with_text_connection(self): + def test_run_with_text_input_connection(self): with self.dataset_populator.test_history() as history_id: self._run_jobs(""" class: GalaxyWorkflow @@ -1996,6 +1996,27 @@ text_input: content = self.dataset_populator.get_history_dataset_content(history_id) self.assertEqual("chrX\t152691446\t152691471\tCCDS14735.1_cds_0_0_chrX_152691447_f\t0\t+\n", content) + def test_run_with_numeric_input_connection(self): + history_id = self.dataset_populator.new_history() + self._run_jobs(""" +class: GalaxyWorkflow +steps: +- label: forty_two + tool_id: expression_forty_two + state: {} +- label: consume_expression_parameter + tool_id: cheetah_casting + state: + floattest: 3.14 + inttest: + $link: forty_two#out1 +test_data: {} +""", history_id=history_id) + + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + content = self.dataset_populator.get_history_dataset_content(history_id) + self.assertEquals("43\n4.14\n", content) + @skip_without_tool('cat1') def test_workflow_rerun_with_use_cached_job(self): workflow = self.workflow_populator.load_workflow(name="test_for_run") diff --git a/test/functional/tools/expression_forty_two.xml b/test/functional/tools/expression_forty_two.xml new file mode 100644 index 00000000000..c30553f85fa --- /dev/null +++ b/test/functional/tools/expression_forty_two.xml @@ -0,0 +1,15 @@ + + Parse Int + + {return {'output': + 42}; + } + + + + + + + Produces the integer 42. + diff --git a/test/functional/tools/expression_log_line_count.xml b/test/functional/tools/expression_log_line_count.xml new file mode 100644 index 00000000000..26804a97127 --- /dev/null +++ b/test/functional/tools/expression_log_line_count.xml @@ -0,0 +1,14 @@ + + Log Lines + + {return {'output': Math.max(Math.round(Math.log(parseInt($job.input1.metadata.data_lines))), 1)};} + + + + + + + + + diff --git a/test/functional/tools/expression_parse_int.xml b/test/functional/tools/expression_parse_int.xml new file mode 100644 index 00000000000..df2e9b98436 --- /dev/null +++ b/test/functional/tools/expression_parse_int.xml @@ -0,0 +1,14 @@ + + Parse Int + + {return {'output': parseInt($job.input1)};} + + + + + + + + Parse an integer from text. + diff --git a/test/functional/tools/samples_tool_conf.xml b/test/functional/tools/samples_tool_conf.xml index 2409bd0dd4b..77d4f11b78d 100644 --- a/test/functional/tools/samples_tool_conf.xml +++ b/test/functional/tools/samples_tool_conf.xml @@ -147,8 +147,10 @@ + + + - diff --git a/test/unit/jobs/test_expression_run.py b/test/unit/jobs/test_expression_run.py new file mode 100644 index 00000000000..71f07a0b97c --- /dev/null +++ b/test/unit/jobs/test_expression_run.py @@ -0,0 +1,48 @@ +import json +import os +import tempfile +import shutil +import subprocess + +from galaxy.tools import expressions + +THIS_DIRECTORY = os.path.abspath(os.path.dirname(__file__)) +TEST_DIRECTORY = os.path.join(THIS_DIRECTORY, os.path.pardir, os.path.pardir) +ROOT_DIRECTORY = os.path.join(TEST_DIRECTORY, os.path.pardir) +LIB_DIRECTORY = os.path.join(ROOT_DIRECTORY, "lib") + + +def test_run_simple(): + test_directory = tempfile.mkdtemp() + try: + environment_path = os.path.join(test_directory, "env.json") + environment = { + 'job': {'input1': '7'}, + 'outputs': [ + {'name': 'out1', 'from_expression': "output1", 'path': 'moo'} + ], + 'script': "{return {'output1': parseInt($job.input1)};}", + } + with open(environment_path, "w") as f: + json.dump(environment, f) + expressions.write_evalute_script( + test_directory, + ) + new_env = os.environ.copy() + if "PYTHONPATH" in new_env: + new_env['PYTHONPATH'] = "%s:%s" % (LIB_DIRECTORY, new_env["PYTHONPATH"]) + else: + new_env['PYTHONPATH'] = "%s" % (LIB_DIRECTORY) + new_env['GALAXY_EXPRESSION_INPUTS'] = environment_path + p = subprocess.Popen( + args=expressions.EXPRESSION_SCRIPT_CALL, + shell=True, + cwd=test_directory, + env=new_env, + ) + assert p.wait() == 0 + with open(os.path.join(test_directory, 'moo')) as f: + out_content = f.read() + assert out_content == '7', out_content + finally: + shutil.rmtree(test_directory) diff --git a/test/unit/tools/test_parsing.py b/test/unit/tools/test_parsing.py index a87b9734e0d..e442cd5ca21 100644 --- a/test/unit/tools/test_parsing.py +++ b/test/unit/tools/test_parsing.py @@ -92,6 +92,39 @@ tests: compare: sim_size """ +TOOL_EXPRESSION_XML_1 = """ + + Parse Int + + {return {'output': parseInt($job.input1)};} + + + + + + + + Parse an integer from text. + +""" + + +TOOL_EXPRESSION_YAML_1 = """ +class: GalaxyExpressionTool +name: "parse_int" +id: parse_int +version: 1.0.2 +expression: "{return {'output': parseInt($job.input1)};}" +inputs: + - name: input1 + label: Text to parse + type: text +outputs: + out1: + type: integer + from: "#output" +""" + class BaseLoaderTestCase(unittest.TestCase): @@ -119,6 +152,22 @@ class BaseLoaderTestCase(unittest.TestCase): return tool_source +class XmlExpressionLoaderTestCase(BaseLoaderTestCase): + source_file_name = "expression.xml" + source_contents = TOOL_EXPRESSION_XML_1 + + def test_expression(self): + assert self._tool_source.parse_expression().strip() == "{return {'output': parseInt($job.input1)};}" + + def test_tool_type(self): + assert self._tool_source.parse_tool_type() == "expression" + + +class YamlExpressionLoaderTestCase(BaseLoaderTestCase): + source_file_name = "expression.yml" + source_contents = TOOL_EXPRESSION_XML_1 + + class XmlLoaderTestCase(BaseLoaderTestCase): source_file_name = "bwa.xml" source_contents = TOOL_XML_1