From fa84dda4884ddd31e06b9cba89ef0c745c8935f5 Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Fri, 19 Jun 2026 21:09:38 +0200 Subject: [PATCH] Verify the custom-tool eval's container is a real biocontainer, not a made-up tag ToolYamlContains can confirm the container is in the quay.io/biocontainers namespace, but a substring match can't distinguish a real image from a hallucinated tag (an invented build suffix). Adds a ContainerVerified evaluator that checks the produced tool's container against quay.io via biocontainer_tag_built: - verified-present biocontainer -> 1.0 - verifiably-absent tag (made up) -> 0.0 - not a biocontainers ref, or a transient lookup failure -> 1.0 (no false failure) Wired into build_custom_tool, so every custom_tool case is scored. Touches the network like the recommender; degrades to 1.0 when outbound calls aren't available. --- test/evals/evaluators.py | 41 ++++++++++++++++++++++++++++++++++++++++ test/evals/specs.py | 4 ++++ 2 files changed, 45 insertions(+) diff --git a/test/evals/evaluators.py b/test/evals/evaluators.py index eabf0adb293..b989a8780b8 100644 --- a/test/evals/evaluators.py +++ b/test/evals/evaluators.py @@ -7,11 +7,14 @@ from typing import ( TypeVar, ) +import yaml from pydantic_evals.evaluators import ( Evaluator, EvaluatorContext, ) +from galaxy.tool_util.deps.mulled.recommend import biocontainer_tag_built + # Routing datasets vary in their case-input shape: a plain query string, or a dict for the # multi-turn cases. HandoffMatch ignores the input entirely (it only compares output to # expected), so it is generic over the input type and adapts to whichever dataset it scores. @@ -145,6 +148,44 @@ class ToolYamlContains(Evaluator[Any, Any, dict]): return hits / len(needles) +@dataclass +class ContainerVerified(Evaluator[Any, Any, dict]): + """Score 0.0 only when the produced tool's container is a *verifiably made-up* + biocontainer; 1.0 otherwise. + + ``ToolYamlContains`` can assert the container is in the ``quay.io/biocontainers`` + namespace, but a substring match can't tell a real image from a hallucinated tag + (e.g. an invented ``--py311h1128e8f_0`` build suffix). This checks the actual + ``container`` against quay.io with ``biocontainer_tag_built``: + + - tag verified present (real biocontainer) -> 1.0 + - tag verified absent (made up) -> 0.0 <- the failure this exists to catch + - unverifiable (not a biocontainers reference, or a transient lookup failure) + -> 1.0, so a legitimately real non-biocontainer image (busybox, ubuntu, ...) + or a network blip is not a false failure. Whether the image *ought* to be a + biocontainer is a separate concern, measured by ``ToolYamlContains``. + + Touches the network (quay.io), like the recommender -- meaningful only in eval + runs where outbound calls are allowed; it degrades to 1.0 when they aren't. + """ + + def evaluate(self, ctx: EvaluatorContext[Any, Any, dict]) -> float: + output = ctx.output if isinstance(ctx.output, dict) else {} + yaml_text = str(output.get("tool_yaml") or "") + if not yaml_text: + return 0.0 + try: + data = yaml.safe_load(yaml_text) + except yaml.YAMLError: + data = None + container = data.get("container") if isinstance(data, dict) else None + if not container: + return 0.0 + # 0.0 only on a positively-disproven biocontainer tag; None (unverifiable) + # and True (present) both pass. + return 0.0 if biocontainer_tag_built(str(container)) is False else 1.0 + + @dataclass class ToolCallMatch(Evaluator[Any, Any, dict]): """Score 1.0 if the model called any tool named in metadata['expected_tool_calls']. diff --git a/test/evals/specs.py b/test/evals/specs.py index 4a3f289dfe8..55bd64e18db 100644 --- a/test/evals/specs.py +++ b/test/evals/specs.py @@ -35,6 +35,7 @@ from .datasets import ( tool_recommendation_dataset, ) from .evaluators import ( + ContainerVerified, FirstAttemptOk, HandoffMatch, MustMention, @@ -270,6 +271,9 @@ def build_custom_tool( dataset.add_evaluator(ToolProduced()) dataset.add_evaluator(FirstAttemptOk()) dataset.add_evaluator(ToolYamlContains()) + # Verify the produced container isn't a hallucinated biocontainer tag (checks + # against quay.io; passes for verified-real, non-biocontainer, or unverifiable). + dataset.add_evaluator(ContainerVerified()) return BuiltDataset( dataset=dataset, task=make_custom_tool_task(deps, usage_buffer=usage_buffer),