diff --git a/config/datatypes_conf.xml.sample b/config/datatypes_conf.xml.sample index 76fdb0c7ee5..12f4fae5d2f 100644 --- a/config/datatypes_conf.xml.sample +++ b/config/datatypes_conf.xml.sample @@ -414,6 +414,7 @@ + diff --git a/lib/galaxy/config.py b/lib/galaxy/config.py index 5717c087a09..d02eb8d99c9 100644 --- a/lib/galaxy/config.py +++ b/lib/galaxy/config.py @@ -313,6 +313,7 @@ class Configuration(object): log.warning("preserve_python_environment set to unknown value [%s], defaulting to legacy_only") preserve_python_environment = "legacy_only" self.preserve_python_environment = preserve_python_environment + self.nodejs_path = kwargs.get("nodejs_path", None) # Older default container cache path, I don't think anyone is using it anymore and it wasn't documented - we # should probably drop the backward compatiblity to save the path check. self.container_image_cache_path = self.resolve_path(kwargs.get("container_image_cache_path", "database/container_images")) diff --git a/lib/galaxy/datatypes/text.py b/lib/galaxy/datatypes/text.py index c2e71444cf1..c563c2d318c 100644 --- a/lib/galaxy/datatypes/text.py +++ b/lib/galaxy/datatypes/text.py @@ -106,6 +106,30 @@ class Json(Text): return "JSON file (%s)" % (nice_size(dataset.get_size())) +class ExpressionJson(Json): + """ Represents the non-data input or output to a tool or workflow. + """ + file_ext = "json" + MetadataElement(name="json_type", default=None, desc="JavaScript or JSON type of expression", readonly=True, visible=True, no_value=None) + + def set_meta(self, dataset, **kwd): + """ + """ + json_type = "null" + with open(dataset.file_name) as f: + obj = json.load(f) + if isinstance(obj, int): + json_type = "int" + elif isinstance(obj, float): + json_type = "float" + elif isinstance(obj, list): + json_type = "list" + elif isinstance(obj, dict): + json_type = "object" + + dataset.metadata.json_type = json_type + + @build_sniff_from_prefix class Ipynb(Json): file_ext = "ipynb" diff --git a/lib/galaxy/dependencies/pipfiles/default/Pipfile b/lib/galaxy/dependencies/pipfiles/default/Pipfile index b6ad3afea6d..abc4b2e76cd 100644 --- a/lib/galaxy/dependencies/pipfiles/default/Pipfile +++ b/lib/galaxy/dependencies/pipfiles/default/Pipfile @@ -40,6 +40,7 @@ uWSGI = "*" pysam = "==0.15.2" bdbag = "==1.4.1" # 1.5.0 requires Python >=2.7.9 bleach = "*" +cwltool = "==1.0.20180721142728" "bz2file" = {version = "*", markers = "python_version < '3.3'"} ipaddress = {version = "*", markers = "python_version < '3.3'"} isa-rwval = "*" diff --git a/lib/galaxy/dependencies/pipfiles/default/pinned-dev-requirements.txt b/lib/galaxy/dependencies/pipfiles/default/pinned-dev-requirements.txt index 0dc921a00ac..fef1dbf7b42 100644 --- a/lib/galaxy/dependencies/pipfiles/default/pinned-dev-requirements.txt +++ b/lib/galaxy/dependencies/pipfiles/default/pinned-dev-requirements.txt @@ -23,7 +23,7 @@ more-itertools==5.0.0 ; python_version <= '2.7' nose==1.3.7 nosehtml==0.4.5 packaging==19.0 -pathlib2==2.3.3 ; python_version < '3.6' +pathlib2==2.3.2 ; python_version < '3' pathtools==0.1.2 pbr==5.1.3 pluggy==0.9.0 diff --git a/lib/galaxy/dependencies/pipfiles/default/pinned-requirements.txt b/lib/galaxy/dependencies/pipfiles/default/pinned-requirements.txt index 1cdfb5c8161..42ed79a78d0 100644 --- a/lib/galaxy/dependencies/pipfiles/default/pinned-requirements.txt +++ b/lib/galaxy/dependencies/pipfiles/default/pinned-requirements.txt @@ -4,6 +4,7 @@ adal==1.2.1 amqp==2.4.2 appdirs==1.4.3 asn1crypto==0.24.0 +avro==1.8.1 ; python_version < '3' azure-common==1.1.14 azure-cosmosdb-nspkg==2.0.2 azure-cosmosdb-table==1.0.4 @@ -31,6 +32,7 @@ botocore==1.12.115 bunch==1.0.1 bx-python==0.8.2 bz2file==0.98 ; python_version < '3.3' +cachecontrol==0.11.7 cachetools==3.1.0 certifi==2019.3.9 cffi==1.12.2 @@ -42,6 +44,7 @@ cloudbridge==2.0.0 cmd2==0.8.9 contextlib2==0.5.5 ; python_version < '3.5' cryptography==2.6.1 +cwltool==1.0.20180721142728 debtcollector==1.21.0 decorator==4.3.2 deprecated==1.2.5 @@ -75,14 +78,18 @@ jsonpointer==2.0 jsonschema==2.6.0 keystoneauth1==3.13.1 kombu==4.4.0 +lockfile==0.12.2 +lxml==4.3.2 mako==1.0.7 markupsafe==1.1.1 mercurial==3.7.3 ; python_version < '3' +mistune==0.8.4 monotonic==1.5 msgpack==0.6.1 msrest==0.5.5 msrestazure==0.5.0 munch==2.3.2 +mypy-extensions==0.4.1 netaddr==0.7.19 netifaces==0.10.9 networkx==1.11 @@ -107,14 +114,17 @@ parsley==1.3 paste==3.0.8 pastedeploy==2.0.1 pastescript==3.1.0 +pathlib2==2.3.2 ; python_version < '3' pbr==5.1.3 prettytable==0.7.2 +prov==1.5.1 psutil==5.6.1 pulsar-galaxy-lib==0.8.3 pyasn1-modules==0.2.4 pyasn1==0.4.5 pycparser==2.19 pycryptodome==3.7.3 +pyeventsystem==0.1.0 pyinotify==0.9.6 ; sys_platform != 'win32' and sys_platform != 'darwin' and sys_platform != 'sunos5' pyjwt==1.7.1 pykwalify==1.7.0 @@ -136,6 +146,8 @@ python-openid==2.2.5 ; python_version < '3.0' python-swiftclient==3.6.0 pytz==2018.9 pyyaml==5.1 +rdflib-jsonld==0.4.0 +rdflib==4.2.2 repoze.lru==0.7 requests-oauthlib==1.2.0 requests-toolbelt==0.9.1 @@ -144,7 +156,12 @@ requestsexceptions==1.4.0 rfc3986==1.2.0 routes==2.4.1 rsa==4.0 +ruamel.ordereddict==0.4.13 ; platform_python_implementation == 'CPython' and python_version <= '2.7' +ruamel.yaml==0.15.89 s3transfer==0.2.0 +scandir==1.10.0 ; python_version < '3.5' +schema-salad==2.7.20181126142424 +shellescape==3.4.1 simplejson==3.16.0 six==1.11.0 social-auth-core[openidconnect]==3.1.0 @@ -157,7 +174,8 @@ subprocess32==3.5.3 ; python_version < '3.0' svgwrite==1.2.1 tempita==0.5.2 tenacity==4.12.0 -typing==3.6.6 ; python_version < '3.5' +typing-extensions==3.7.2 +typing==3.6.6 ; python_version < '3.6' tzlocal==1.5.1 unicodecsv==0.14.1 ; python_version < '3.0' uritemplate==3.0.0 diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index 4e6af300f94..7e22d209b4d 100755 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -29,6 +29,7 @@ from galaxy.managers.jobs import JobSearch from galaxy.metadata import get_metadata_compute_strategy from galaxy.model.tags import GalaxyTagHandler from galaxy.queue_worker import send_control_task +from galaxy.tools import expressions from galaxy.tools.actions import DefaultToolAction from galaxy.tools.actions.data_manager import DataManagerToolAction from galaxy.tools.actions.data_source import DataSourceToolAction @@ -62,6 +63,7 @@ from galaxy.tools.parameters.dataset_matcher import ( from galaxy.tools.parameters.grouping import Conditional, ConditionalWhen, Repeat, Section, UploadDataset from galaxy.tools.parameters.input_translation import ToolInputTranslator from galaxy.tools.parameters.meta import expand_meta_parameters +from galaxy.tools.parameters.wrapped_json import json_wrap from galaxy.tools.parser import ( get_tool_source, get_tool_source_from_representation, @@ -103,6 +105,9 @@ from .provided_metadata import parse_tool_provided_metadata log = logging.getLogger(__name__) +REQUIRES_JS_RUNTIME_MESSAGE = ("The tool [%s] requires a nodejs runtime to execute " + "but node or nodejs could not be found. Please contact the Galaxy adminstrator") + HELP_UNINITIALIZED = threading.Lock() MODEL_TOOLS_PATH = os.path.abspath(os.path.dirname(__file__)) # Tools that require Galaxy's Python environment to be preserved. @@ -778,6 +783,12 @@ class Tool(Dictifiable): module, cls = action mod = __import__(module, globals(), locals(), [cls]) self.tool_action = getattr(mod, cls)() + if getattr(self.tool_action, "requires_js_runtime", False): + try: + expressions.find_engine(self.app.config) + except Exception: + message = REQUIRES_JS_RUNTIME_MESSAGE % self.tool_id or self.tool_uuid + raise Exception(message) # Tests self.__parse_tests(tool_source) @@ -2209,6 +2220,72 @@ class OutputParameterJSONTool(Tool): out.close() +class ExpressionTool(Tool): + requires_js_runtime = True + tool_type = 'expression' + EXPRESSION_INPUTS_NAME = "_expression_inputs_.json" + + def parse_command(self, tool_source): + self.command = "cd ../; %s" % expressions.EXPRESSION_SCRIPT_CALL + self.interpreter = None + self._expression = tool_source.parse_expression().strip() + + def parse_outputs(self, tool_source): + # Setup self.outputs and self.output_collections + super(ExpressionTool, self).parse_outputs(tool_source) + + # Validate these outputs for expression tools. + if len(self.output_collections) != 0: + message = "Expression tools may not declare output collections at this time." + raise Exception(message) + for output in self.outputs.values(): + if not hasattr(output, "from_expression"): + message = "Expression tools may not declare output datasets at this time." + raise Exception(message) + + def exec_before_job(self, app, inp_data, out_data, param_dict=None): + super(ExpressionTool, self).exec_before_job(app, inp_data, out_data, param_dict=param_dict) + local_working_directory = param_dict["__local_working_directory__"] + expression_inputs_path = os.path.join(local_working_directory, ExpressionTool.EXPRESSION_INPUTS_NAME) + + outputs = [] + for i, (out_name, data) in enumerate(out_data.items()): + output_def = self.outputs[out_name] + wrapped_data = param_dict.get(out_name) + file_name = str(wrapped_data) + + outputs.append(dict( + name=out_name, + from_expression=output_def.from_expression, + path=file_name, + )) + + if param_dict is None: + raise Exception("Internal error - param_dict is empty.") + + job = {} + json_wrap(self.inputs, param_dict, job, handle_files='OBJECT') + expression_inputs = { + 'job': job, + 'script': self._expression, + 'outputs': outputs, + } + expressions.write_evalute_script(os.path.join(local_working_directory)) + with open(expression_inputs_path, "w") as f: + json.dump(expression_inputs, f) + + def parse_environment_variables(self, tool_source): + """ Setup environment variable for inputs file. + """ + environmnt_variables_raw = super(ExpressionTool, self).parse_environment_variables(tool_source) + expression_script_inputs = dict( + name="GALAXY_EXPRESSION_INPUTS", + template=ExpressionTool.EXPRESSION_INPUTS_NAME, + ) + environmnt_variables_raw.append(expression_script_inputs) + return environmnt_variables_raw + + class DataSourceTool(OutputParameterJSONTool): """ Alternate implementation of Tool for data_source tools -- those that @@ -2955,7 +3032,7 @@ class FilterFromFileTool(DatabaseOperationTool): # Populate tool_type to ToolClass mappings tool_types = {} -for tool_class in [Tool, SetMetadataTool, OutputParameterJSONTool, +for tool_class in [Tool, SetMetadataTool, OutputParameterJSONTool, ExpressionTool, DataManagerTool, DataSourceTool, AsyncDataSourceTool, UnzipCollectionTool, ZipCollectionTool, MergeCollectionTool, RelabelFromFileTool, FilterFromFileTool, BuildListCollectionTool, ExtractDatasetCollectionTool, diff --git a/lib/galaxy/tools/evaluation.py b/lib/galaxy/tools/evaluation.py index b244364ac6e..9e56df5e56c 100644 --- a/lib/galaxy/tools/evaluation.py +++ b/lib/galaxy/tools/evaluation.py @@ -380,7 +380,7 @@ class ToolEvaluator(object): param_dict['__tool_directory__'] = self.compute_environment.tool_directory() param_dict['__get_data_table_entry__'] = get_data_table_entry - + param_dict['__local_working_directory__'] = self.local_working_directory # We add access to app here, this allows access to app.config, etc param_dict['__app__'] = RawObjectWrapper(self.app) # More convienent access to app.config.new_file_path; we don't need to diff --git a/lib/galaxy/tools/expressions/__init__.py b/lib/galaxy/tools/expressions/__init__.py new file mode 100644 index 00000000000..bf20411a392 --- /dev/null +++ b/lib/galaxy/tools/expressions/__init__.py @@ -0,0 +1,16 @@ +from .evaluation import evaluate +from .script import ( + EXPRESSION_SCRIPT_CALL, + EXPRESSION_SCRIPT_NAME, + write_evalute_script, +) +from .util import find_engine + + +__all__ = ( + 'evaluate', + 'EXPRESSION_SCRIPT_CALL', + 'EXPRESSION_SCRIPT_NAME', + 'find_engine', + 'write_evalute_script', +) diff --git a/lib/galaxy/tools/expressions/cwlNodeEngine.js b/lib/galaxy/tools/expressions/cwlNodeEngine.js new file mode 100644 index 00000000000..1129584977f --- /dev/null +++ b/lib/galaxy/tools/expressions/cwlNodeEngine.js @@ -0,0 +1,46 @@ +#!/usr/bin/env nodejs + +"use strict"; + +process.stdin.setEncoding('utf8'); + +var incoming = ""; + +process.stdin.on('readable', function() { + var chunk = process.stdin.read(); + if (chunk !== null) { + incoming += chunk; + } +}); + +process.stdin.on('end', function() { + var j = JSON.parse(incoming); + var exp = "" + + if (j.script[0] == "{") { + exp = "{return function()" + j.script + "();}"; + } + else { + exp = "{return " + j.script + ";}"; + } + + var fn = '"use strict";\n'; + + if (j.engineConfig) { + for (var index = 0; index < j.engineConfig.length; ++index) { + fn += j.engineConfig[index] + "\n"; + } + } + + fn += "var $job = " + JSON.stringify(j.job) + ";\n"; + fn += "var $self = " + JSON.stringify(j.context) + ";\n" + + fn += "var $runtime = " + JSON.stringify(j.runtime) + ";\n" + fn += "var $tmpdir = " + JSON.stringify(j.tmpdir) + ";\n" + fn += "var $outdir = " + JSON.stringify(j.outdir) + ";\n" + + + fn += "(function()" + exp + ")()"; + + process.stdout.write(JSON.stringify(require("vm").runInNewContext(fn, {}))); +}); diff --git a/lib/galaxy/tools/expressions/evaluation.py b/lib/galaxy/tools/expressions/evaluation.py new file mode 100644 index 00000000000..54066d5094b --- /dev/null +++ b/lib/galaxy/tools/expressions/evaluation.py @@ -0,0 +1,38 @@ +import json +import os +import subprocess + +from .util import find_engine + +FILE_DIRECTORY = os.path.normpath(os.path.dirname(os.path.join(__file__))) +NODE_ENGINE = os.path.join(FILE_DIRECTORY, "cwlNodeEngine.js") + + +def evaluate(config, input): + application = find_engine(config) + + default_context = { + "engineConfig": [], + "job": {}, + "context": None, + "outdir": None, + "tmpdir": None, + } + + new_input = default_context + new_input.update(input) + + sp = subprocess.Popen([application, NODE_ENGINE], + shell=False, + close_fds=True, + stdin=subprocess.PIPE, + stdout=subprocess.PIPE) + input_str = json.dumps(new_input) + "\n\n" + input_bytes = input_str.encode("utf-8") + (stdoutdata, stderrdata) = sp.communicate(input_bytes) + if sp.returncode != 0: + args = (json.dumps(new_input, indent=4), stdoutdata, stderrdata) + message = "Expression engine returned non-zero exit code on evaluation of\n%s%s%s" % args + raise Exception(message) + + return json.loads(stdoutdata.decode("utf-8")) diff --git a/lib/galaxy/tools/expressions/script.py b/lib/galaxy/tools/expressions/script.py new file mode 100644 index 00000000000..0cb18ae7464 --- /dev/null +++ b/lib/galaxy/tools/expressions/script.py @@ -0,0 +1,15 @@ +import os + +EXPRESSION_SCRIPT_NAME = "_evaluate_expression_.py" +EXPRESSION_SCRIPT_CALL = "python %s" % EXPRESSION_SCRIPT_NAME + + +def write_evalute_script(in_directory): + """ Responsible for writing the script that evaluates expressions + in Galaxy jobs. + """ + script = os.path.join(in_directory, EXPRESSION_SCRIPT_NAME) + with open(script, "w") as f: + f.write('from galaxy_ext.expressions.handle_job import run; run()') + + return script diff --git a/lib/galaxy/tools/expressions/util.py b/lib/galaxy/tools/expressions/util.py new file mode 100644 index 00000000000..082ed3c916f --- /dev/null +++ b/lib/galaxy/tools/expressions/util.py @@ -0,0 +1,10 @@ +from galaxy.tools.deps.commands import which + + +def find_engine(config): + nodejs_path = getattr(config, "nodejs_path", None) + if nodejs_path is None: + nodejs_path = which("nodejs") or which("node") or None + if nodejs_path is None: + raise Exception("nodejs or node not found on PATH") + return nodejs_path diff --git a/lib/galaxy/tools/parameters/wrapped_json.py b/lib/galaxy/tools/parameters/wrapped_json.py index c045c9595ef..f9c90d4abe1 100644 --- a/lib/galaxy/tools/parameters/wrapped_json.py +++ b/lib/galaxy/tools/parameters/wrapped_json.py @@ -62,6 +62,13 @@ def _json_wrap_input(input, value, handle_files="skip"): json_value = _data_input_to_path(value) elif handle_files == "skip": return SKIP_INPUT + elif handle_files == "OBJECT": + if value: + if isinstance(value, list): + value = value[0] + return _hda_to_object(value) + else: + return None else: raise NotImplementedError() elif input_type == "data_collection": @@ -88,6 +95,21 @@ def _json_wrap_input(input, value, handle_files="skip"): return json_value +def _hda_to_object(hda): + hda_dict = hda.to_dict() + metadata_dict = {} + + for key, value in hda_dict.items(): + if key.startswith("metadata_"): + metadata_dict[key[len("metadata_"):]] = value + + return { + 'file_ext': hda_dict['file_ext'], + 'name': hda_dict['name'], + 'metadata': metadata_dict, + } + + def _cast_if_not_none(value, cast_to, empty_to_none=False): # log.debug("value [%s], type[%s]" % (value, type(value))) if value is None or (empty_to_none and str(value) == ''): diff --git a/lib/galaxy/tools/parser/interface.py b/lib/galaxy/tools/parser/interface.py index 2cd31a1aa6f..bdc979dccfc 100644 --- a/lib/galaxy/tools/parser/interface.py +++ b/lib/galaxy/tools/parser/interface.py @@ -88,6 +88,11 @@ class ToolSource(object): """ Return string contianing command to run. """ + def parse_expression(self): + """ Return string contianing command to run. + """ + return None + @abstractmethod def parse_environment_variables(self): """ Return environment variable templates to expose. diff --git a/lib/galaxy/tools/parser/output_objects.py b/lib/galaxy/tools/parser/output_objects.py index bfe51934ff9..805335a8358 100644 --- a/lib/galaxy/tools/parser/output_objects.py +++ b/lib/galaxy/tools/parser/output_objects.py @@ -24,12 +24,13 @@ class ToolOutput(ToolOutputBase): (format, metadata_source, parent) """ - dict_collection_visible_keys = ['name', 'format', 'label', 'hidden'] + dict_collection_visible_keys = ['name', 'format', 'label', 'hidden', 'output_type'] def __init__(self, name, format=None, format_source=None, metadata_source=None, parent=None, label=None, filters=None, actions=None, hidden=False, implicit=False): super(ToolOutput, self).__init__(name, label=label, filters=filters, hidden=hidden) + self.output_type = "data" self.format = format self.format_source = format_source self.metadata_source = metadata_source @@ -70,6 +71,27 @@ class ToolOutput(ToolOutputBase): return as_dict +class ToolExpressionOutput(ToolOutputBase): + dict_collection_visible_keys = ('name', 'format', 'label', 'hidden', 'output_type') + + def __init__(self, name, output_type, from_expression, + label=None, filters=None, actions=None, hidden=False): + super(ToolExpressionOutput, self).__init__(name, label=label, filters=filters, hidden=hidden) + self.output_type = output_type # JSON type... + self.from_expression = from_expression + self.format = "expression.json" # galaxy.datatypes.text.ExpressionJson.file_ext + + self.format_source = None + self.metadata_source = None + self.parent = None + self.actions = actions + + # Initialize default values + self.change_format = [] + self.implicit = False + self.from_work_dir = None + + class ToolOutputCollection(ToolOutputBase): """ Represents a HistoryDatasetCollectionAssociation of output datasets produced @@ -85,6 +107,7 @@ class ToolOutputCollection(ToolOutputBase): """ + dict_collection_visible_keys = ('name', 'format', 'label', 'hidden', 'output_type') dict_collection_visible_keys = ['name', 'default_format', 'label', 'hidden', 'inherit_format', 'inherit_metadata'] @@ -102,6 +125,7 @@ class ToolOutputCollection(ToolOutputBase): inherit_metadata=False ): super(ToolOutputCollection, self).__init__(name, label=label, filters=filters, hidden=hidden) + self.output_type = "collection" self.collection = True self.default_format = default_format self.structure = structure diff --git a/lib/galaxy/tools/parser/xml.py b/lib/galaxy/tools/parser/xml.py index 997eb760406..c142c97745d 100644 --- a/lib/galaxy/tools/parser/xml.py +++ b/lib/galaxy/tools/parser/xml.py @@ -22,6 +22,7 @@ from .interface import ( from .output_actions import ToolOutputActionGroup from .output_collection_def import dataset_collector_descriptions_from_elem from .output_objects import ( + ToolExpressionOutput, ToolOutput, ToolOutputCollection, ToolOutputCollectionStructure @@ -111,6 +112,17 @@ class XmlToolSource(ToolSource): command_el = self._command_el return ((command_el is not None) and command_el.text) or None + def parse_expression(self): + """ Return string contianing command to run. + """ + expression_el = self.root.find("expression") + if expression_el is not None: + expression_type = expression_el.get("type") + if expression_type != "ecma5.1": + raise Exception("Unknown expression type [%s] encountered" % expression_type) + return expression_el.text + return None + def parse_environment_variables(self): environment_variables_el = self.root.find("environment_variables") if environment_variables_el is None: @@ -256,7 +268,12 @@ class XmlToolSource(ToolSource): for _ in out_elem.findall("data"): _parse(_) - for collection_elem in out_elem.findall("collection"): + def _parse_expression(output_elem, **kwds): + output_def = self._parse_expression_output(output_elem, tool, **kwds) + data_dict[output_def.name] = output_def + return output_def + + def _parse_collection(collection_elem): name = collection_elem.get("name") label = xml_text(collection_elem, "label") default_format = collection_elem.get("format", "data") @@ -312,6 +329,24 @@ class XmlToolSource(ToolSource): output_collection.outputs[output_name] = data output_collections[name] = output_collection + for out_child in out_elem.getchildren(): + if out_child.tag == "data": + _parse(out_child) + elif out_child.tag == "collection": + _parse_collection(out_child) + elif out_child.tag == "output": + output_type = out_child.get("type") + if output_type == "data": + _parse(out_child) + elif output_type == "collection": + out_child.attrib["type"] = out_child.get("collection_type") + out_child.attrib["type_source"] = out_child.get("collection_type_source") + _parse_collection(out_child) + else: + _parse_expression(out_child) + else: + log.warn("Unknown output tag encountered [%s]" % out_child.tag) + for output_def in data_dict.values(): outputs[output_def.name] = output_def return outputs, output_collections @@ -323,6 +358,7 @@ class XmlToolSource(ToolSource): default_format="data", default_format_source=None, default_metadata_source="", + expression_type=None, ): output = ToolOutput(data_elem.get("name")) output_format = data_elem.get("format", default_format) @@ -347,6 +383,22 @@ class XmlToolSource(ToolSource): output.dataset_collector_descriptions = dataset_collector_descriptions_from_elem(data_elem, legacy=self.legacy_defaults) return output + def _parse_expression_output(self, output_elem, tool, **kwds): + output_type = output_elem.get("type") + from_expression = output_elem.get("from") + output = ToolExpressionOutput( + output_elem.get("name"), + output_type, + from_expression, + ) + output.path = output_elem.get("value") + output.label = xml_text(output_elem, "label") + + output.hidden = string_as_bool(output_elem.get("hidden", "")) + output.actions = ToolOutputActionGroup(output, output_elem.find('actions')) + output.dataset_collector_descriptions = [] + return output + def parse_stdio(self): """ parse error handling from command and stdio tag diff --git a/lib/galaxy/tools/parser/yaml.py b/lib/galaxy/tools/parser/yaml.py index 63e642e5648..52ae9d4cb07 100644 --- a/lib/galaxy/tools/parser/yaml.py +++ b/lib/galaxy/tools/parser/yaml.py @@ -54,6 +54,9 @@ class YamlToolSource(ToolSource): def parse_command(self): return self.root_dict.get("command") + def parse_expression(self): + return self.root_dict.get("expression") + def parse_environment_variables(self): return [] diff --git a/lib/galaxy/tools/xsd/galaxy.xsd b/lib/galaxy/tools/xsd/galaxy.xsd index 17e3191df80..0f59437410f 100644 --- a/lib/galaxy/tools/xsd/galaxy.xsd +++ b/lib/galaxy/tools/xsd/galaxy.xsd @@ -67,7 +67,8 @@ the tool menu immediately following the hyperlink for the tool (based on the - + + @@ -2781,6 +2782,33 @@ Prior to Galaxy release 19.01 the stdio block has only been used for non-legacy + + + + + + + + + Type of expression defined by this expression block. The only current valid option is ecma5.1 - which will evaluate the expression in a sandbox using node. The option still must be specified to allow a different default in the future. + + + + + + + + + + + + + + @@ -3684,6 +3712,7 @@ The default is ``galaxy.json``. + @@ -3725,6 +3754,71 @@ pipes or periods (e.g. ``.``).]]> + + + + + + + + + The short name for the output datatype. +The valid values for format can be found in +[/config/datatypes_conf.xml.sample](https://github.com/galaxyproject/galaxy/blob/dev/config/datatypes_conf.xml.sample) +(e.g. ``format="pdf"`` or ``format="fastqsanger"``). + + + + + Sets the source of element identifier to the specified input. +This only applies to collections that are mapped over a non-collection input and that have equivalent structures. If this references input elements in conditionals, this value should be qualified (e.g. ``cond|input`` instead of ``input`` if ``input`` is in a conditional with ``name="cond"``). + + + + + This copies the metadata information +from the tool's input dataset. This is particularly useful for interval data +types where the order of the columns is not set. + + + + + Relative path to a file produced by the +tool in its working directory. Output's contents are set to this file's +contents. + + + + + Boolean indicating whether to hide +dataset in the history view. (Default is ``false``.) + + + + + + + + This is the name of input collection or +dataset to derive "structure" of the output from (output element count and +identifiers). For instance, if the referenced input has three ordered items with +identifiers ``sample1``, ``sample2``, and ``sample3``. If this references input +elements in conditionals, this value should be qualified (e.g. ``cond|input`` instead +of ``input`` if ``input`` is in a conditional with ``name="cond"``). + + + + + If ``structured_like`` is set, inherit +format of outputs from format of corresponding input. + + + + - - - - - - - - The short name for the output datatype. -The valid values for format can be found in -[/config/datatypes_conf.xml.sample](https://github.com/galaxyproject/galaxy/blob/dev/config/datatypes_conf.xml.sample) -(e.g. ``format="pdf"`` or ``format="fastqsanger"``). - - - - - Sets the source of element identifier to the specified input. -This only applies to collections that are mapped over a non-collection input and that have equivalent structures. If this references input elements in conditionals, this value should be qualified (e.g. ``cond|input`` instead of ``input`` if ``input`` is in a conditional with ``name="cond"``). - - - - - This copies the metadata information -from the tool's input dataset. This is particularly useful for interval data -types where the order of the columns is not set. - - - - - Relative path to a file produced by the -tool in its working directory. Output's contents are set to this file's -contents. - - - - - Boolean indicating whether to hide -dataset in the history view. (Default is ``false``.) - - + @@ -3887,6 +3939,7 @@ Creating collections in tools is covered in-depth in + Collection type for output (e.g. ``paired``, ``list``, or ``list:list``). @@ -3898,24 +3951,44 @@ Creating collections in tools is covered in-depth in derive collection's type (e.g. ``collection_type``) from. - - - This is the name of input collection or -dataset to derive "structure" of the output from (output element count and -identifiers). For instance, if the referenced input has three ordered items with -identifiers ``sample1``, ``sample2``, and ``sample3``. If this references input -elements in conditionals, this value should be qualified (e.g. ``cond|input`` instead -of ``input`` if ``input`` is in a conditional with ``name="cond"``). - - - - - If ``structured_like`` is set, inherit -format of outputs from format of corresponding input. - - + + + + + + + + + + + + + Output type. This could be older more established Galaxy types (e.g. data and collection) - in which case the semantics of this largely reflect the corresponding ``data`` and ``collection`` tags. This could also be newer non-data types such as ``integer`` or ``boolean``. + + + + + In expression tools, use this to specify a dictionary value to populate this output from. The semantics may change for other expression types in the future. + + + + + Collection type for output (e.g. ``paired``, ``list``, or ``list:list``). + + + + + This is the name of input collection to +derive collection's type (e.g. ``collection_type``) from. + + + + + diff --git a/lib/galaxy/workflow/modules.py b/lib/galaxy/workflow/modules.py index f87e164025c..6c1cf39ba1f 100644 --- a/lib/galaxy/workflow/modules.py +++ b/lib/galaxy/workflow/modules.py @@ -45,6 +45,7 @@ from galaxy.tools.parameters.basic import ( ) from galaxy.tools.parameters.history_query import HistoryQuery from galaxy.tools.parameters.wrapped import make_dict_copy +from galaxy.tools.parser.output_objects import ToolExpressionOutput from galaxy.util.bunch import Bunch from galaxy.util.json import safe_loads from galaxy.util.odict import odict @@ -961,6 +962,8 @@ class ToolModule(WorkflowModule): if filter_output(tool_output, self.state.inputs): continue extra_kwds = {} + if isinstance(tool_output, ToolExpressionOutput): + extra_kwds['parameter'] = True if tool_output.collection: extra_kwds["collection"] = True collection_type = tool_output.structure.collection_type @@ -997,6 +1000,7 @@ class ToolModule(WorkflowModule): dict( name=name, extensions=formats, + type=tool_output.output_type, **extra_kwds ) ) diff --git a/lib/galaxy/workflow/run.py b/lib/galaxy/workflow/run.py index b75bee7ecac..961a673e9f4 100644 --- a/lib/galaxy/workflow/run.py +++ b/lib/galaxy/workflow/run.py @@ -1,3 +1,4 @@ +import json import logging import uuid @@ -367,6 +368,17 @@ class WorkflowProgress(object): delayed_why = "dependent collection [%s] not yet populated with datasets" % replacement.id raise modules.DelayedWorkflowEvaluation(why=delayed_why) + + is_hda = isinstance(replacement, model.HistoryDatasetAssociation) + if not is_data and is_hda: + if replacement.is_ok: + with open(replacement.file_name, 'r') as f: + replacement = json.load(f) + elif replacement.is_pending: + raise modules.DelayedWorkflowEvaluation() + else: + raise modules.CancelWorkflowEvaluation() + return replacement def get_replacement_workflow_output(self, workflow_output): diff --git a/lib/galaxy_ext/expressions/__init__.py b/lib/galaxy_ext/expressions/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/lib/galaxy_ext/expressions/handle_job.py b/lib/galaxy_ext/expressions/handle_job.py new file mode 100644 index 00000000000..680e16fad20 --- /dev/null +++ b/lib/galaxy_ext/expressions/handle_job.py @@ -0,0 +1,45 @@ +""" +Execute an external process to evaluate expressions for Galaxy jobs. + +Galaxy should be importable on sys.path . +""" +import json +import logging +import os +import sys + +# insert *this* galaxy before all others on sys.path +sys.path.insert(1, os.path.abspath(os.path.join(os.path.dirname(__file__), os.pardir, os.pardir))) + +try: + from cwltool import expression +except ImportError: + expression = None + +from galaxy.tools.expressions import evaluate + +logging.basicConfig() +log = logging.getLogger(__name__) + + +def run(environment_path=None): + if expression is None: + raise Exception("Python library cwltool must be available to evaluate expressions.") + + if environment_path is None: + environment_path = os.environ.get("GALAXY_EXPRESSION_INPUTS") + with open(environment_path, "r") as f: + raw_inputs = json.load(f) + + outputs = raw_inputs["outputs"] + inputs = raw_inputs.copy() + del inputs["outputs"] + + result = evaluate(None, inputs) + + for output in outputs: + path = output["path"] + from_expression = "$(" + output["from_expression"] + ")" + output_value = expression.interpolate(from_expression, result) + with open(path, "w") as f: + json.dump(output_value, f) diff --git a/scripts/common_startup.sh b/scripts/common_startup.sh index 8306473ab86..65a9545536f 100755 --- a/scripts/common_startup.sh +++ b/scripts/common_startup.sh @@ -224,14 +224,23 @@ if [ $SKIP_CLIENT_BUILD -eq 0 ]; then fi fi +# Install node if not installed +if [ -n "$VIRTUAL_ENV" ]; then + if ! in_venv "$(command -v node)" || [ "$(node --version)" != "v$NODE_VERSION" ]; then + echo "Installing node into $VIRTUAL_ENV with nodeenv." + nodeenv -n $NODE_VERSION -p + fi +elif [ -n "$CONDA_DEFAULT_ENV" -a -n "$CONDA_EXE" ]; then + if ! in_conda_env "$(command -v node)"; then + echo "Installing node into '$CONDA_DEFAULT_ENV' Conda environment with conda." + $CONDA_EXE install --yes --override-channels --channel conda-forge --channel defaults --name $CONDA_DEFAULT_ENV node=$NODE_VERSION + fi +fi + # Build client if necessary. if [ $SKIP_CLIENT_BUILD -eq 0 ]; then # Ensure dependencies are installed if [ -n "$VIRTUAL_ENV" ]; then - if ! in_venv "$(command -v node)" || [ "$(node --version)" != "v$NODE_VERSION" ]; then - echo "Installing node into $VIRTUAL_ENV with nodeenv." - nodeenv -n $NODE_VERSION -p - fi if ! in_venv "$(command -v yarn)"; then echo "Installing yarn into $VIRTUAL_ENV with npm." npm install --global yarn diff --git a/test/api/test_tools.py b/test/api/test_tools.py index 011a5652716..60f2cd52c35 100644 --- a/test/api/test_tools.py +++ b/test/api/test_tools.py @@ -2130,6 +2130,48 @@ class ToolsTestCase(api.ApiTestCase): output_content = self.dataset_populator.get_history_dataset_content(history_id, dataset=output) self.assertEqual(output_content.strip(), "123\n456\n456\n0ab") + @skip_without_tool("expression_forty_two") + def test_galaxy_expression_tool_simplest(self): + history_id = self.dataset_populator.new_history() + inputs = { + } + run_response = self._run( + "expression_forty_two", history_id, inputs + ) + self._assert_status_code_is(run_response, 200) + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + output_content = self.dataset_populator.get_history_dataset_content(history_id) + self.assertEqual(output_content, "42") + + @skip_without_tool("expression_parse_int") + def test_galaxy_expression_tool_simple(self): + history_id = self.dataset_populator.new_history() + inputs = { + 'input1': '7', + } + run_response = self._run( + "expression_parse_int", history_id, inputs + ) + self._assert_status_code_is(run_response, 200) + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + output_content = self.dataset_populator.get_history_dataset_content(history_id) + self.assertEqual(output_content, "7") + + @skip_without_tool("expression_log_line_count") + def test_galaxy_expression_metadata(self): + history_id = self.dataset_populator.new_history() + new_dataset1 = self.dataset_populator.new_dataset(history_id, content='1\n2\n3\n4\n5\n6\n7\n8\n9\n10\n11\n12\n13\n14') + inputs = { + 'input1': dataset_to_param(new_dataset1), + } + run_response = self._run( + "expression_log_line_count", history_id, inputs + ) + self._assert_status_code_is(run_response, 200) + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + output_content = self.dataset_populator.get_history_dataset_content(history_id) + self.assertEqual(output_content, "3") + def __build_group_list(self, history_id): response = self.dataset_collection_populator.upload_collection(history_id, "list", elements=[ { diff --git a/test/api/test_workflows.py b/test/api/test_workflows.py index 246e8be341c..73b6c3b6e75 100644 --- a/test/api/test_workflows.py +++ b/test/api/test_workflows.py @@ -822,12 +822,14 @@ test_data: """, history_id=history_id, assert_ok=False, wait=False) self.wait_for_invocation_and_jobs(history_id, job_summary.workflow_id, job_summary.invocation_id, assert_ok=False) history_contents = self.dataset_populator._get_contents_request(history_id=history_id).json() + first_input = history_contents[1] + assert first_input['history_content_type'] == 'dataset' paused_dataset = history_contents[-1] failed_dataset = self.dataset_populator.get_history_dataset_details(history_id, hid=5, assert_ok=False) assert paused_dataset['state'] == 'paused', paused_dataset assert failed_dataset['state'] == 'error', failed_dataset inputs = {"input1": {'values': [{'src': 'hda', - 'id': history_contents[0]['id']}] + 'id': first_input['id']}] }, "failbool": "false", "rerun_remap_job_id": failed_dataset['creating_job']} @@ -1996,7 +1998,7 @@ text_input: type: raw """, history_id=history_id, wait=True, assert_ok=False) - def test_run_with_text_connection(self): + def test_run_with_text_input_connection(self): with self.dataset_populator.test_history() as history_id: self._run_jobs(""" class: GalaxyWorkflow @@ -2027,6 +2029,27 @@ text_input: content = self.dataset_populator.get_history_dataset_content(history_id) self.assertEqual("chrX\t152691446\t152691471\tCCDS14735.1_cds_0_0_chrX_152691447_f\t0\t+\n", content) + def test_run_with_numeric_input_connection(self): + history_id = self.dataset_populator.new_history() + self._run_jobs(""" +class: GalaxyWorkflow +steps: +- label: forty_two + tool_id: expression_forty_two + state: {} +- label: consume_expression_parameter + tool_id: cheetah_casting + state: + floattest: 3.14 + inttest: + $link: forty_two#out1 +test_data: {} +""", history_id=history_id) + + self.dataset_populator.wait_for_history(history_id, assert_ok=True) + content = self.dataset_populator.get_history_dataset_content(history_id) + self.assertEquals("43\n4.14\n", content) + @skip_without_tool('cat1') def test_workflow_rerun_with_use_cached_job(self): workflow = self.workflow_populator.load_workflow(name="test_for_run") diff --git a/test/functional/tools/expression_forty_two.xml b/test/functional/tools/expression_forty_two.xml new file mode 100644 index 00000000000..af6afd0d997 --- /dev/null +++ b/test/functional/tools/expression_forty_two.xml @@ -0,0 +1,15 @@ + + Parse Int + + {return {'output': + 42}; + } + + + + + + + Produces the integer 42. + diff --git a/test/functional/tools/expression_log_line_count.xml b/test/functional/tools/expression_log_line_count.xml new file mode 100644 index 00000000000..b5811769906 --- /dev/null +++ b/test/functional/tools/expression_log_line_count.xml @@ -0,0 +1,14 @@ + + Log Lines + + {return {'output': Math.max(Math.round(Math.log(parseInt($job.input1.metadata.data_lines))), 1)};} + + + + + + + + + diff --git a/test/functional/tools/expression_parse_int.xml b/test/functional/tools/expression_parse_int.xml new file mode 100644 index 00000000000..3e2080aac0c --- /dev/null +++ b/test/functional/tools/expression_parse_int.xml @@ -0,0 +1,14 @@ + + Parse Int + + {return {'output': parseInt($job.input1)};} + + + + + + + + Parse an integer from text. + diff --git a/test/functional/tools/samples_tool_conf.xml b/test/functional/tools/samples_tool_conf.xml index 2409bd0dd4b..77d4f11b78d 100644 --- a/test/functional/tools/samples_tool_conf.xml +++ b/test/functional/tools/samples_tool_conf.xml @@ -147,8 +147,10 @@ + + + - diff --git a/test/unit/jobs/test_expression_run.py b/test/unit/jobs/test_expression_run.py new file mode 100644 index 00000000000..c23455d6e07 --- /dev/null +++ b/test/unit/jobs/test_expression_run.py @@ -0,0 +1,48 @@ +import json +import os +import shutil +import subprocess +import tempfile + +from galaxy.tools import expressions + +THIS_DIRECTORY = os.path.abspath(os.path.dirname(__file__)) +TEST_DIRECTORY = os.path.join(THIS_DIRECTORY, os.path.pardir, os.path.pardir) +ROOT_DIRECTORY = os.path.join(TEST_DIRECTORY, os.path.pardir) +LIB_DIRECTORY = os.path.join(ROOT_DIRECTORY, "lib") + + +def test_run_simple(): + test_directory = tempfile.mkdtemp() + try: + environment_path = os.path.join(test_directory, "env.json") + environment = { + 'job': {'input1': '7'}, + 'outputs': [ + {'name': 'out1', 'from_expression': "output1", 'path': 'moo'} + ], + 'script': "{return {'output1': parseInt($job.input1)};}", + } + with open(environment_path, "w") as f: + json.dump(environment, f) + expressions.write_evalute_script( + test_directory, + ) + new_env = os.environ.copy() + if "PYTHONPATH" in new_env: + new_env['PYTHONPATH'] = "%s:%s" % (LIB_DIRECTORY, new_env["PYTHONPATH"]) + else: + new_env['PYTHONPATH'] = "%s" % (LIB_DIRECTORY) + new_env['GALAXY_EXPRESSION_INPUTS'] = environment_path + p = subprocess.Popen( + args=expressions.EXPRESSION_SCRIPT_CALL, + shell=True, + cwd=test_directory, + env=new_env, + ) + assert p.wait() == 0 + with open(os.path.join(test_directory, 'moo')) as f: + out_content = f.read() + assert out_content == '7', out_content + finally: + shutil.rmtree(test_directory) diff --git a/test/unit/tools/test_expression_basics.py b/test/unit/tools/test_expression_basics.py new file mode 100644 index 00000000000..51c7293ac3f --- /dev/null +++ b/test/unit/tools/test_expression_basics.py @@ -0,0 +1,8 @@ +from galaxy.tools.expressions import evaluate + + +def test_evaluate(): + input = { + "script": "{return 5;}" + } + assert evaluate(None, input) == 5 diff --git a/test/unit/tools/test_parsing.py b/test/unit/tools/test_parsing.py index a87b9734e0d..772a4b98418 100644 --- a/test/unit/tools/test_parsing.py +++ b/test/unit/tools/test_parsing.py @@ -92,6 +92,39 @@ tests: compare: sim_size """ +TOOL_EXPRESSION_XML_1 = """ + + Parse Int + + {return {'output': parseInt($job.input1)};} + + + + + + + + Parse an integer from text. + +""" + + +TOOL_EXPRESSION_YAML_1 = """ +class: GalaxyExpressionTool +name: "parse_int" +id: parse_int +version: 1.0.2 +expression: "{return {'output': parseInt($job.input1)};}" +inputs: + - name: input1 + label: Text to parse + type: text +outputs: + out1: + type: integer + from: "#output" +""" + class BaseLoaderTestCase(unittest.TestCase): @@ -119,6 +152,22 @@ class BaseLoaderTestCase(unittest.TestCase): return tool_source +class XmlExpressionLoaderTestCase(BaseLoaderTestCase): + source_file_name = "expression.xml" + source_contents = TOOL_EXPRESSION_XML_1 + + def test_expression(self): + assert self._tool_source.parse_expression().strip() == "{return {'output': parseInt($job.input1)};}" + + def test_tool_type(self): + assert self._tool_source.parse_tool_type() == "expression" + + +class YamlExpressionLoaderTestCase(BaseLoaderTestCase): + source_file_name = "expression.yml" + source_contents = TOOL_EXPRESSION_XML_1 + + class XmlLoaderTestCase(BaseLoaderTestCase): source_file_name = "bwa.xml" source_contents = TOOL_XML_1 diff --git a/test/unit/workflows/test_modules.py b/test/unit/workflows/test_modules.py index 921596b8d9f..4ad43a127f2 100644 --- a/test/unit/workflows/test_modules.py +++ b/test/unit/workflows/test_modules.py @@ -452,7 +452,8 @@ def __mock_tool( format_source=None, change_format=[], filters=[], - label=None)}, + label=None, + output_type='data')}, params_from_strings=mock.Mock(), check_and_update_param_values=mock.Mock(), to_json=_to_json