Add strict Pydantic Job model for TestJob.job

Follow-up to #18884 which modeled TestJob.outputs but left TestJob.job
as Dict[str, Any]. Schema defines the canonical CWL-style workflow-test
input syntax; legacy type: File|Directory|raw forms stay in populator
helpers and are no longer accepted in fixtures.

- New galaxy.tool_util_models.test_job with Job = RootModel[dict[str, ...]]
  - File variants (LocationFile, PathFile, ContentsFile, CompositeDataFile)
    dispatched via callable Discriminator on dict-shape
  - Collection with recursive CollectionElement, strict collection_type
  - Directory (supported by galactic_job_json; IWC unused but in-tree
    bwa_mem2_index fixture needs it)
  - HashEntry using galaxy.util.hash_util.HashFunctionNames
  - Non-recursive list axis (List[File|scalar]) — no observed deeper nesting
- Moved StrictModel + CollectionType + _check_collection_type to _base.py
  so test_job.py can reuse them without circular import.
- TestJob.job: Dict[str, Any] -> Job; Job re-exported at package top.
- test_test_format_model.py: 12 positive + 7 negative parametrized unit
  fixtures covering every union arm and the canonical rejection cases.
- replacement_parameters_legacy.gxwf-tests.yml parked on skip list with
  explanation — replacement_parameters: {...} is Planemo-era magic popped
  by run_workflow, not a canonical job input.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
John Chilton
2026-04-22 09:53:59 -04:00
co-authored by Claude Opus 4.7
parent abd7a9a5b7
commit aef25d3cb0
24 changed files with 395 additions and 21 deletions
+10 -21
View File
@@ -13,7 +13,6 @@ from typing import (
)
from pydantic import (
AfterValidator,
AnyUrl,
BaseModel,
ConfigDict,
@@ -28,9 +27,14 @@ from typing_extensions import (
TypedDict,
)
from ._base import ToolSourceBaseModel
from ._base import (
CollectionType,
StrictModel,
ToolSourceBaseModel,
)
from .assertions import assertions
from .parameters import ToolParameterT
from .test_job import Job
from .tool_outputs import (
IncomingToolOutput,
ToolOutput,
@@ -165,11 +169,6 @@ class ParsedTool(ToolSourceBaseModel):
help: Optional[HelpContent]
class StrictModel(BaseModel):
model_config = ConfigDict(extra="forbid", field_title_generator=lambda field_name, field_info: field_name.lower())
class BaseTestOutputModel(StrictModel):
file: Optional[str] = None
path: Optional[str] = None
@@ -205,19 +204,6 @@ TestCollectionElementAssertion = Union[
TestCollectionCollectionElementAssertions.model_rebuild()
def _check_collection_type(v: str) -> str:
if len(v) == 0:
raise ValueError("Invalid empty collection_type specified.")
collection_levels = v.split(":")
for collection_level in collection_levels:
if collection_level not in ["list", "paired", "paired_or_unpaired", "record", "sample_sheet"]:
raise ValueError(f"Invalid collection_type specified [{v}]")
return v
CollectionType = Annotated[Optional[str], AfterValidator(_check_collection_type)]
class CollectionAttributes(StrictModel):
collection_type: CollectionType = None
@@ -307,12 +293,15 @@ class YamlToolTest(BaseModel):
UserToolSource.model_rebuild()
YamlToolSource.model_rebuild()
# Loose alias retained for TestJobDict / TypedDict consumers where helpers
# still tolerate Dict[str, Any]. The strict, validated shape is `Job` (see
# galaxy.tool_util_models.test_job).
JobDict = Dict[str, Any]
class TestJob(StrictModel):
doc: Optional[str]
job: JobDict
job: Job
outputs: Dict[str, TestOutputAssertions]
expect_failure: Optional[bool] = False
+20
View File
@@ -1,10 +1,30 @@
"""Base model classes for tool utilities."""
from typing import Optional
from pydantic import (
AfterValidator,
BaseModel,
ConfigDict,
)
from typing_extensions import Annotated
class ToolSourceBaseModel(BaseModel):
model_config = ConfigDict(field_title_generator=lambda field_name, field_info: field_name.lower())
class StrictModel(BaseModel):
model_config = ConfigDict(extra="forbid", field_title_generator=lambda field_name, field_info: field_name.lower())
def _check_collection_type(v: str) -> str:
if len(v) == 0:
raise ValueError("Invalid empty collection_type specified.")
for level in v.split(":"):
if level not in ("list", "paired", "paired_or_unpaired", "record", "sample_sheet"):
raise ValueError(f"Invalid collection_type specified [{v}]")
return v
CollectionType = Annotated[Optional[str], AfterValidator(_check_collection_type)]
+182
View File
@@ -0,0 +1,182 @@
"""Pydantic model for the ``job:`` block of Planemo / ``*.gxwf-tests.yml`` tests.
Defines the canonical CWL-style workflow-test input syntax — the shape the
schema blesses, not every shape the helpers in ``galaxy_test.base.populators``
happen to tolerate. Legacy ``type: File | Directory | raw`` / ``value`` forms
are intentionally not modeled; those are left to helper-layer tolerances and
should not leak into test fixtures.
Follow-up to galaxyproject/galaxy#18884, which modeled ``TestJob.outputs`` but
left ``TestJob.job`` as ``Dict[str, Any]``.
"""
from typing import (
Optional,
Union,
)
from pydantic import (
ConfigDict,
Discriminator,
Field,
RootModel,
Tag,
)
from typing_extensions import (
Annotated,
Literal,
)
from ._base import (
CollectionType,
StrictModel,
)
# Mirrored from galaxy.util.hash_util.HashFunctionNames — duplicated rather than
# imported so this package stays free of a galaxy-util runtime dep. Kept in sync
# by test/unit/tool_util_models/test_hash_function_names_sync.py, which runs in
# the monorepo where galaxy.util is available.
HashFunctionNames = Literal["MD5", "SHA-1", "SHA-256", "SHA-512"]
class _StrictJobModel(StrictModel):
model_config = ConfigDict(
extra="forbid",
populate_by_name=True,
field_title_generator=lambda field_name, field_info: field_name.lower(),
)
class HashEntry(_StrictJobModel):
hash_function: HashFunctionNames
hash_value: str
class BaseFile(_StrictJobModel):
"""Fields common to every ``class: File`` variant."""
class_: Literal["File"] = Field(alias="class")
filetype: Optional[str] = None
dbkey: Optional[str] = None
decompress: Optional[bool] = None
to_posix_lines: Optional[bool] = None
space_to_tab: Optional[bool] = None
deferred: Optional[bool] = None
name: Optional[str] = None
info: Optional[str] = None
tags: Optional[list[str]] = None
hashes: Optional[list[HashEntry]] = None
identifier: Optional[str] = None
class LocationFile(BaseFile):
location: str
path: Optional[str] = None
contents: Optional[str] = None
composite_data: Optional[list[str]] = None
class PathFile(BaseFile):
path: str
location: Optional[str] = None
contents: Optional[str] = None
composite_data: Optional[list[str]] = None
class ContentsFile(BaseFile):
"""CWL File literal — content inlined as a string."""
contents: str
path: Optional[str] = None
location: Optional[str] = None
composite_data: Optional[list[str]] = None
class CompositeDataFile(BaseFile):
composite_data: list[str]
path: Optional[str] = None
location: Optional[str] = None
contents: Optional[str] = None
def _discriminate_file(v):
if isinstance(v, dict):
if "location" in v:
return "location"
if "path" in v:
return "path"
if "contents" in v:
return "contents"
if "composite_data" in v:
return "composite_data"
return None
if isinstance(v, LocationFile):
return "location"
if isinstance(v, PathFile):
return "path"
if isinstance(v, ContentsFile):
return "contents"
if isinstance(v, CompositeDataFile):
return "composite_data"
return None
File = Annotated[
Union[
Annotated[LocationFile, Tag("location")],
Annotated[PathFile, Tag("path")],
Annotated[ContentsFile, Tag("contents")],
Annotated[CompositeDataFile, Tag("composite_data")],
],
Discriminator(_discriminate_file),
]
class Collection(_StrictJobModel):
class_: Literal["Collection"] = Field(alias="class")
collection_type: CollectionType = None
name: Optional[str] = None
identifier: Optional[str] = None
elements: Optional[list["CollectionElement"]] = None
rows: Optional[dict[str, list]] = None
CollectionElement = Annotated[
Union[File, Collection],
Field(discriminator="class_"),
]
Collection.model_rebuild()
class Directory(_StrictJobModel):
"""CWL-style directory input. Supported by stage_inputs for directory-typed
datasets (e.g. bwa_mem2_index test fixtures). Rare in workflow tests; IWC
does not use it.
"""
class_: Literal["Directory"] = Field(alias="class")
path: Optional[str] = None
location: Optional[str] = None
filetype: Optional[str] = None
name: Optional[str] = None
# JobParamValue is non-recursive at the list axis: a job-param list may contain
# files or scalars but not further nested lists. Collection nesting (lists of
# collections) is recursive via ``CollectionElement``. No observed workflow
# test value needs a list-of-lists at the job-input level; widen explicitly if
# that changes rather than defaulting to Any.
JobParamValue = Union[
File,
Collection,
Directory,
str,
int,
float,
bool,
None,
list[Union[File, str, int, float, bool, None]],
]
Job = RootModel[dict[str, JobParamValue]]
@@ -0,0 +1,10 @@
input:
class: Collection
collection_type: list
elements:
- identifier: el1
class: File
path: a.txt
- identifier: el2
class: File
path: b.txt
@@ -0,0 +1,13 @@
input:
class: Collection
collection_type: list:paired
elements:
- identifier: p1
class: Collection
elements:
- identifier: forward
class: File
contents: "F"
- identifier: reverse
class: File
contents: "R"
@@ -0,0 +1,10 @@
input:
class: Collection
collection_type: paired
elements:
- identifier: forward
class: File
contents: "forward content"
- identifier: reverse
class: File
contents: "reverse content"
@@ -0,0 +1,4 @@
reference:
class: Directory
path: bwa_mem2_index
filetype: bwa_mem2_index
@@ -0,0 +1,6 @@
input:
class: File
filetype: imzml
composite_data:
- Example_Continuous.imzML
- Example_Continuous.ibd
@@ -0,0 +1,4 @@
input:
class: File
contents: "inline file content\n"
filetype: txt
@@ -0,0 +1,7 @@
input:
class: File
location: https://example.org/test.fasta
filetype: fasta
hashes:
- hash_function: SHA-256
hash_value: e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855
@@ -0,0 +1,4 @@
input:
class: File
path: 1.fasta
filetype: fasta
@@ -0,0 +1,14 @@
input:
class: File
path: 1.fasta
filetype: fasta
dbkey: hg19
decompress: true
to_posix_lines: true
space_to_tab: false
deferred: false
name: reference-genome
info: sample info
tags:
- group:project
- name:reference
@@ -0,0 +1,5 @@
multi_data:
- class: File
path: a.txt
- class: File
path: b.txt
@@ -0,0 +1,4 @@
multi_text:
- --ex1
- --ex2
- --ex3
@@ -0,0 +1,4 @@
input:
class: Collection
collection_type: banana
elements: []
@@ -0,0 +1,6 @@
input:
collection_type: list
elements:
- identifier: el1
class: File
path: a.txt
@@ -0,0 +1,3 @@
input:
class: File
filetype: fasta
@@ -0,0 +1,4 @@
input:
type: File
value: 1.fasta
file_type: fasta
@@ -0,0 +1,3 @@
input:
type: raw
value: 5
@@ -0,0 +1 @@
"bare string at top level of job block"
@@ -0,0 +1,4 @@
input:
class: File
path: 1.fasta
flargh: 1
@@ -0,0 +1,5 @@
text_input: hello
int_input: 42
float_input: 3.14
bool_input: true
null_input: null
@@ -0,0 +1,14 @@
"""Guard that the mirrored HashFunctionNames in tool_util_models stays in sync
with galaxy.util.hash_util. The mirror exists so the package has no runtime
dep on galaxy.util; this test lives here (not in test/unit/tool_util_models/)
so it runs in the monorepo env where galaxy.util is importable.
"""
from typing import get_args
from galaxy.tool_util_models.test_job import HashFunctionNames as MirroredHashFunctionNames
from galaxy.util.hash_util import HashFunctionNames
def test_hash_function_names_in_sync():
assert get_args(MirroredHashFunctionNames) == get_args(HashFunctionNames)
@@ -2,20 +2,78 @@ import os
from pathlib import Path
from typing import List
import pytest
import yaml
from pydantic import ValidationError
from galaxy.tool_util_models import Tests
from galaxy.tool_util_models.test_job import Job
from galaxy.util import galaxy_directory
from galaxy.util.unittest_utils import skip_unless_environ
TEST_WORKFLOW_DIRECTORY = os.path.join(galaxy_directory(), "lib", "galaxy_test", "workflow")
IWC_WORKFLOWS_USING_UNVERIFIED_SYNTAX: List[str] = []
# replacement_parameters_legacy.gxwf-tests.yml is a deliberate regression test
# for Planemo-era implicit replacement_parameters: {...} dicts embedded in
# job:. That key is popped out by WorkflowPopulator.run_workflow before
# staging, so the runtime accepts it but it is not canonical workflow-test
# input syntax and the strict Job schema does not model it.
WORKFLOW_TESTS_SKIP_STRICT_VALIDATION: List[str] = [
"replacement_parameters_legacy.gxwf-tests.yml",
]
TEST_JOB_FIXTURES_DIR = Path(__file__).parent / "test_data" / "test_job_fixtures"
def _load_fixture(name: str):
with open(TEST_JOB_FIXTURES_DIR / name) as f:
return yaml.safe_load(f)
JOB_POSITIVE_FIXTURES = [
"file_path.yml",
"file_location.yml",
"file_composite.yml",
"file_contents.yml",
"file_with_tags_and_dbkey.yml",
"collection_list.yml",
"collection_paired.yml",
"collection_nested.yml",
"directory.yml",
"scalars.yml",
"list_of_files.yml",
"list_of_scalars.yml",
]
JOB_NEGATIVE_FIXTURES = [
"neg_legacy_type_file.yml",
"neg_legacy_type_raw.yml",
"neg_elements_without_class.yml",
"neg_file_no_path_or_location.yml",
"neg_unknown_field.yml",
"neg_bad_collection_type.yml",
"neg_top_level_not_dict.yml",
]
@pytest.mark.parametrize("fixture", JOB_POSITIVE_FIXTURES)
def test_job_positive_fixtures(fixture: str) -> None:
Job.model_validate(_load_fixture(fixture))
@pytest.mark.parametrize("fixture", JOB_NEGATIVE_FIXTURES)
def test_job_negative_fixtures(fixture: str) -> None:
with pytest.raises(ValidationError):
Job.model_validate(_load_fixture(fixture))
def test_validate_workflow_tests():
path = Path(TEST_WORKFLOW_DIRECTORY)
test_files = path.glob("*.gxwf-tests.yml")
for test_file in test_files:
if test_file.name in WORKFLOW_TESTS_SKIP_STRICT_VALIDATION:
continue
with open(test_file) as f:
json = yaml.safe_load(f)
Tests.model_validate(json)