Verify the custom-tool eval's container is a real biocontainer, not a made-up tag

ToolYamlContains can confirm the container is in the quay.io/biocontainers
namespace, but a substring match can't distinguish a real image from a hallucinated
tag (an invented build suffix). Adds a ContainerVerified evaluator that checks the
produced tool's container against quay.io via biocontainer_tag_built:

- verified-present biocontainer -> 1.0
- verifiably-absent tag (made up) -> 0.0
- not a biocontainers ref, or a transient lookup failure -> 1.0 (no false failure)

Wired into build_custom_tool, so every custom_tool case is scored. Touches the
network like the recommender; degrades to 1.0 when outbound calls aren't available.
This commit is contained in:
mvdbeek
2026-07-07 16:39:18 +02:00
parent 87f0cbc83e
commit fa84dda488
2 changed files with 45 additions and 0 deletions
+41
View File
@@ -7,11 +7,14 @@ from typing import (
TypeVar,
)
import yaml
from pydantic_evals.evaluators import (
Evaluator,
EvaluatorContext,
)
from galaxy.tool_util.deps.mulled.recommend import biocontainer_tag_built
# Routing datasets vary in their case-input shape: a plain query string, or a dict for the
# multi-turn cases. HandoffMatch ignores the input entirely (it only compares output to
# expected), so it is generic over the input type and adapts to whichever dataset it scores.
@@ -145,6 +148,44 @@ class ToolYamlContains(Evaluator[Any, Any, dict]):
return hits / len(needles)
@dataclass
class ContainerVerified(Evaluator[Any, Any, dict]):
"""Score 0.0 only when the produced tool's container is a *verifiably made-up*
biocontainer; 1.0 otherwise.
``ToolYamlContains`` can assert the container is in the ``quay.io/biocontainers``
namespace, but a substring match can't tell a real image from a hallucinated tag
(e.g. an invented ``--py311h1128e8f_0`` build suffix). This checks the actual
``container`` against quay.io with ``biocontainer_tag_built``:
- tag verified present (real biocontainer) -> 1.0
- tag verified absent (made up) -> 0.0 <- the failure this exists to catch
- unverifiable (not a biocontainers reference, or a transient lookup failure)
-> 1.0, so a legitimately real non-biocontainer image (busybox, ubuntu, ...)
or a network blip is not a false failure. Whether the image *ought* to be a
biocontainer is a separate concern, measured by ``ToolYamlContains``.
Touches the network (quay.io), like the recommender -- meaningful only in eval
runs where outbound calls are allowed; it degrades to 1.0 when they aren't.
"""
def evaluate(self, ctx: EvaluatorContext[Any, Any, dict]) -> float:
output = ctx.output if isinstance(ctx.output, dict) else {}
yaml_text = str(output.get("tool_yaml") or "")
if not yaml_text:
return 0.0
try:
data = yaml.safe_load(yaml_text)
except yaml.YAMLError:
data = None
container = data.get("container") if isinstance(data, dict) else None
if not container:
return 0.0
# 0.0 only on a positively-disproven biocontainer tag; None (unverifiable)
# and True (present) both pass.
return 0.0 if biocontainer_tag_built(str(container)) is False else 1.0
@dataclass
class ToolCallMatch(Evaluator[Any, Any, dict]):
"""Score 1.0 if the model called any tool named in metadata['expected_tool_calls'].
+4
View File
@@ -35,6 +35,7 @@ from .datasets import (
tool_recommendation_dataset,
)
from .evaluators import (
ContainerVerified,
FirstAttemptOk,
HandoffMatch,
MustMention,
@@ -270,6 +271,9 @@ def build_custom_tool(
dataset.add_evaluator(ToolProduced())
dataset.add_evaluator(FirstAttemptOk())
dataset.add_evaluator(ToolYamlContains())
# Verify the produced container isn't a hallucinated biocontainer tag (checks
# against quay.io; passes for verified-real, non-biocontainer, or unverifiable).
dataset.add_evaluator(ContainerVerified())
return BuiltDataset(
dataset=dataset,
task=make_custom_tool_task(deps, usage_buffer=usage_buffer),