Merge pull request #11063 from bernt-matthias/topic/validate-collections

add validation for collection parameters
This commit is contained in:
Marius van den Beek
2021-01-12 16:40:06 +01:00
committed by GitHub
4 changed files with 162 additions and 70 deletions
+54 -3
View File
@@ -3504,10 +3504,61 @@ target file.</xs:documentation>
<xs:documentation xml:lang="en"><![CDATA[
See the
This tag set is contained within the ``<param>`` tag set - it applies a
validator to the containing parameter. Tool submission will fail if a
single validator fails. See the
[annotation_profiler](https://github.com/galaxyproject/tools-devteam/blob/master/tools/annotation_profiler/annotation_profiler.xml)
tool for an example of how to use this tag set. This tag set is contained within
the ``<param>`` tag set - it applies a validator to the containing parameter.
tool for an example of how to use this tag set.
Note that validators for parameters with ``optional="true"`` are not
executed if no value is given.
### Generic validators
- ``expression``: Check if a one line python expression given expression
evaluates to True. The expression is given is the content of the validator tag.
### Validators for ``data`` and ``data_collection`` parameters
In case of ``data_collection`` parameters and
``data`` parameters with ``multiple="true"`` these validators are executed
separately for each of the contained data sets. Note that, for ``data``
parameters a ``metadata`` validator is added automatically.
- ``metadata``: Check for missing metadata.
- ``unspecified_build``: Check of a build is defined.
- ``dataset_ok_validator``: Check if the data set is in state OK.
- ``dataset_metadata_in_range``: Check if a numeric metadata value is within
a given range.
- ``dataset_metadata_in_file``: Check if a metadata value is contained in a
specific column of another data set.
- ``dataset_metadata_in_data_table`` (``dataset_metadata_not_in_data_table``):
Check if a metadata value is contained in a column of a data table.
### Validators for textual inputs (``text``, ``select``, ...)
``regex``: Check if a regular expression **matches** the value, i.e. appears
at the beginning of the value. To enforce a match of the complete value use
``$`` at the end of the expression. The expression is given is the content
of the validator tag. Note that for ``selects`` each option is checked
separately.
For selects (in particular with dynamically defined options) the following
validator is useful:
``no_options``: Check if options are available for a ``select`` parameter.
Useful for parameters with dynamically defined options.
For ``text`` inputs the following validators are useful:
``length``: Check if the length of the value is within a range.
``empty_field``: Check if the sting is not empty
``value_in_data_table`` (``value_not_in_data_table``): Check if the value is
contained in a column of a given data table.
### Validators for numeric inputs (``integer``, ``float``)
``in_range``: Check if the value is in a given range.
### Examples
+49 -51
View File
@@ -1619,6 +1619,18 @@ class BaseDataToolParameter(ToolParameter):
def __init__(self, tool, input_source, trans):
super().__init__(tool, input_source)
self.min = input_source.get('min')
self.max = input_source.get('max')
if self.min:
try:
self.min = int(self.min)
except ValueError:
raise ParameterValueError("attribute 'min' must be an integer", self.name)
if self.max:
try:
self.max = int(self.max)
except ValueError:
raise ParameterValueError("attribute 'max' must be an integer", self.name)
self.refresh_on_change = True
# Find datatypes_registry
if self.tool is None:
@@ -1747,6 +1759,43 @@ class BaseDataToolParameter(ToolParameter):
else:
return app.model.context.query(app.model.HistoryDatasetAssociation).get(int(value))
def validate(self, value, trans=None):
def do_validate(v):
for validator in self.validators:
if validator.requires_dataset_metadata and v and hasattr(v, 'dataset') and v.dataset.state != galaxy.model.Dataset.states.OK:
return
else:
validator.validate(v, trans)
dataset_count = 0
if value:
if self.multiple:
if not isinstance(value, list):
value = [value]
else:
value = [value]
for v in value:
if isinstance(v, galaxy.model.HistoryDatasetCollectionAssociation):
for dataset_instance in v.collection.dataset_instances:
dataset_count += 1
do_validate(dataset_instance)
elif isinstance(v, galaxy.model.DatasetCollectionElement):
for dataset_instance in v.child_collection.dataset_instances:
dataset_count += 1
do_validate(dataset_instance)
else:
dataset_count += 1
do_validate(v)
if self.min is not None:
if self.min > dataset_count:
raise ValueError("At least %d datasets are required for %s" % (self.min, self.name))
if self.max is not None:
if self.max < dataset_count:
raise ValueError("At most %d datasets are required for %s" % (self.max, self.name))
class DataToolParameter(BaseDataToolParameter):
# TODO, Nate: Make sure the following unit tests appropriately test the dataset security
@@ -1770,18 +1819,6 @@ class DataToolParameter(BaseDataToolParameter):
self.validators.append(validation.MetadataValidator())
self._parse_formats(trans, input_source)
self.multiple = input_source.get_bool('multiple', False)
self.min = input_source.get('min')
self.max = input_source.get('max')
if self.min:
try:
self.min = int(self.min)
except ValueError:
raise ParameterValueError("attribute 'min' must be an integer", self.name)
if self.max:
try:
self.max = int(self.max)
except ValueError:
raise ParameterValueError("attribute 'max' must be an integer", self.name)
if not self.multiple and (self.min is not None):
raise ParameterValueError("cannot specify 'min' property on single data parameter. Set multiple=\"true\" to enable this option", self.name)
if not self.multiple and (self.max is not None):
@@ -1905,42 +1942,6 @@ class DataToolParameter(BaseDataToolParameter):
pass
return "No dataset."
def validate(self, value, trans=None):
dataset_count = 0
for validator in self.validators:
def do_validate(v):
if validator.requires_dataset_metadata and v and hasattr(v, 'dataset') and v.dataset.state != galaxy.model.Dataset.states.OK:
return
else:
validator.validate(v, trans)
if value and self.multiple:
if not isinstance(value, list):
value = [value]
for v in value:
if isinstance(v, galaxy.model.HistoryDatasetCollectionAssociation):
for dataset_instance in v.collection.dataset_instances:
dataset_count += 1
do_validate(dataset_instance)
elif isinstance(v, galaxy.model.DatasetCollectionElement):
for dataset_instance in v.child_collection.dataset_instances:
dataset_count += 1
do_validate(dataset_instance)
else:
dataset_count += 1
do_validate(v)
else:
if value:
dataset_count += 1
do_validate(value)
if self.min is not None:
if self.min > dataset_count:
raise ValueError("At least %d datasets are required for %s" % (self.min, self.name))
if self.max is not None:
if self.max < dataset_count:
raise ValueError("At most %d datasets are required for %s" % (self.max, self.name))
def get_dependencies(self):
"""
Get the *names* of the other params this param depends on.
@@ -2175,9 +2176,6 @@ class DataCollectionToolParameter(BaseDataToolParameter):
display_text = "No dataset collection."
return display_text
def validate(self, value, trans=None):
return True # TODO
def to_dict(self, trans, other_values=None):
# create dictionary and fill default parameters
other_values = other_values or {}
View File
@@ -1,17 +1,60 @@
<tool id="validation_empty_dataset" name="validation_empty_dataset" version="1.0.0">
<command><![CDATA[
echo 'Hello World' > out1
]]></command>
<inputs>
<param name="input1" type="data" label="non-empty input">
<validator type="empty_dataset" />
</param>
</inputs>
<outputs>
<data name="out_file1" from_work_dir="out1" />
</outputs>
<tests>
</tests>
<help>
</help>
<tool id="validation_empty_dataset" name="validation_empty_dataset">
<command>
echo "Hello World" > out1
</command>
<inputs>
<param name="input1" type="data" format="txt" optional="true" label="non-empty input">
<validator type="empty_dataset" />
</param>
<param name="input_mult" type="data" multiple="true" format="txt" optional="true" label="non-empty input">
<validator type="empty_dataset" />
</param>
<param name="input_collection" type="data_collection" format="txt" collection_type="list:paired" optional="true" label="non-empty input">
<validator type="empty_dataset" />
</param>
</inputs>
<outputs>
<data name="out_file1" from_work_dir="out1" format="txt"/>
</outputs>
<tests>
<test expect_failure="true">
<param name="input1" value="empty.txt" ftype="txt"/>
</test>
<test expect_failure="true">
<param name="input_mult" value="1.bed,empty.txt" ftype="txt"/>
</test>
<test expect_failure="true">
<param name="input_collection">
<collection type="list:paired">
<element name="i1">
<collection type="paired">
<element name="forward" value="1.bed" ftype="txt"/>
<element name="reverse" value="empty.txt" ftype="txt"/>
</collection>
</element>
</collection>
</param>
</test>
<test>
<param name="input1" value="1.bed" ftype="txt"/>
<param name="input_mult" value="1.bed,2.bed" ftype="txt"/>
<param name="input_collection">
<collection type="list:paired">
<element name="i1">
<collection type="paired">
<element name="forward" value="1.bed" ftype="txt"/>
<element name="reverse" value="2.bed" ftype="txt"/>
</collection>
</element>
</collection>
</param>
<output name="out_file1">
<assert_contents>
<has_text text="Hello World"/>
</assert_contents>
</output>
</test>
</tests>
<help>
</help>
</tool>