diff --git a/test/evals/README.md b/test/evals/README.md index 42de3e13f94..e4102b8571c 100644 --- a/test/evals/README.md +++ b/test/evals/README.md @@ -32,14 +32,14 @@ Current datasets: galaxyproject/galaxy#21661 (comment 4367167981) where the router answered "what tools are installed?" with a generic essay instead of calling `search_tools`. -- **live26_demo**: canonical prompts from the GCC2026 Live26 demo script - (histological staining quantification flow ending with Omero export). - Scored by `LLMJudge` against per-case rubrics for response substance. - The routing decision for the same prompts is scored separately by the - `live26_*` cases in the `routing` dataset, so a full demo flight check - runs both. Cases needing a live Galaxy session (history sanity check, - save-to-page) are off by default; pass `--include-galaxy-required` to - include them. +- **staining_quantification**: end-to-end bioimaging use case -- + brightfield RGB inputs, color deconvolution, per-ROI quantification, + Omero export. Scored by `LLMJudge` against per-case rubrics for + response substance. The routing decision for the same prompts is + scored separately by the matching cases in the `routing` dataset, so + a full flight check runs both. Cases needing a live Galaxy session + (history sanity check, save-to-page) are off by default; pass + `--include-galaxy-required` to include them. ## Layout @@ -78,8 +78,8 @@ default. `test/integration/test_live_evals.py` runs the same datasets inside a Galaxy integration-test fixture with a real `trans`. Seeds a demo -history via `test/evals/seed_live26_demo_history.py`, runs the -`requires_galaxy=True` cases against it, writes a report to +history via `test/evals/seed_staining_quantification_history.py`, runs +the `requires_galaxy=True` cases against it, writes a report to `test/evals/results/` in the same shape as the CLI. Slower (Galaxy startup), but the only path that actually exercises history-dependent cases. @@ -88,7 +88,7 @@ cases. - Iterating on a prompt? **CLI.** - Choosing between models? **CLI.** -- Stage rehearsal / flight check for the GCC2026 demo? **Pytest live +- End-to-end flight check before a demo rehearsal? **Pytest live runner.** - Cases involving "my history", "my analysis", or the history agent doing real tool calls? **Pytest live runner.** @@ -123,7 +123,7 @@ export GALAXY_TEST_AI_MODEL=gpt-oss-120b # Optional: override which datasets/models/judge to run # export EVALS_MODEL_CONFIG=/path/to/models.yaml -# export EVALS_DATASETS=live26_demo +# export EVALS_DATASETS=staining_quantification # export EVALS_MODELS=gpt-oss-120b # export EVALS_JUDGE_MODEL=gpt-oss-120b @@ -131,9 +131,10 @@ pytest test/integration/test_live_evals.py -v ``` The live runner always passes `include_galaxy_required=True`, so the -history-needing live26 cases (`history_sanity_check`, `summarize_to_page`, -`report_takeaway`, `social_media_post`) actually get exercised. Default -scope is `live26_demo` only; override with `EVALS_DATASETS`. +history-needing staining-quantification cases (`history_sanity_check`, +`summarize_to_page`, `report_takeaway`, `social_media_post`) actually +get exercised. Default scope is `staining_quantification` only; +override with `EVALS_DATASETS`. The default judge is `Llama-4-Maverick-17B-128E-Instruct` rather than `gpt-oss-120b` because gpt-oss-120b tends to grade itself too diff --git a/test/evals/datasets/__init__.py b/test/evals/datasets/__init__.py index 2e798306c63..29af054271a 100644 --- a/test/evals/datasets/__init__.py +++ b/test/evals/datasets/__init__.py @@ -2,18 +2,18 @@ from .bioinformatics_workflows import bioinformatics_workflows_dataset from .error_analysis import error_analysis_dataset -from .live26_demo import live26_demo_dataset from .orchestrator_planning import orchestrator_planning_dataset from .router_tool_use import router_tool_use_dataset from .routing import routing_dataset +from .staining_quantification import staining_quantification_dataset from .tool_recommendation import tool_recommendation_dataset __all__ = [ "bioinformatics_workflows_dataset", "error_analysis_dataset", - "live26_demo_dataset", "orchestrator_planning_dataset", "router_tool_use_dataset", "routing_dataset", + "staining_quantification_dataset", "tool_recommendation_dataset", ] diff --git a/test/evals/datasets/routing.py b/test/evals/datasets/routing.py index 14baf8beab6..4484d27f146 100644 --- a/test/evals/datasets/routing.py +++ b/test/evals/datasets/routing.py @@ -177,50 +177,51 @@ ROUTING_CASES: list[Case[str, str, dict[str, Any]]] = [ "router", "One-word query -- router asks for clarification", ), - # Live26 (GCC2026) demo prompts -- canonical strings from the demo script. - # See evals/datasets/live26_demo.py for content-quality rubrics on the same - # set; these cases only score the routing decision. + # Staining quantification use-case prompts -- a representative + # histological staining quantification flow ending with Omero export. + # See evals/datasets/staining_quantification.py for content-quality + # rubrics on the same set; these cases only score the routing decision. _case( - "live26_stain_quantification_intro", + "stain_quantification_intro", ( "The datasets in my history are brightfield RGB images from a " "histological staining experiment. I'd like to quantify stain " "components from those images. What's a good way to do this?" ), "tool_recommendation", - "Live26 step 3 -- Diana's opening prompt; tool_recommendation also surfaces IWC workflows once agent-ops-iwc-reintroduce lands.", + "Opening prompt; tool_recommendation also surfaces IWC workflows once agent-ops-iwc-reintroduce lands.", ), _case( - "live26_import_iwc_workflow", + "import_iwc_workflow", "Import a histological staining workflow from IWC.", "router", - "Live26 step 4 -- router-direct action. On agent-ops-iwc-reintroduce this triggers search_iwc_workflows + import_workflow_from_iwc tool calls.", + "Router-direct action. On agent-ops-iwc-reintroduce this triggers search_iwc_workflows + import_workflow_from_iwc tool calls.", ), _case( - "live26_omero_upload_guidance", + "omero_upload_guidance", "How can I upload this data to Omero?", "router", - "Live26 step 7 -- router-direct guidance on Omero file source / connection setup.", + "Router-direct guidance on Omero file source / connection setup.", ), _case( - "live26_history_sanity_check", + "history_sanity_check", "Look at my history -- did I miss anything in this analysis?", "history", - "Live26 step 6 -- post-run sanity check.", + "Post-run sanity check on the user's history.", requires_galaxy=True, ), _case( - "live26_summarize_to_page", + "summarize_to_page", "Summarize this analysis and save it as a Galaxy Page.", "history", - "Live26 step 6 follow-up -- history agent owns Page creation.", + "History agent owns Page creation.", requires_galaxy=True, ), _case( - "live26_custom_tool_quantify_brown", + "custom_tool_quantify_brown", "Generate a Galaxy tool that counts brown pixels in a TIFF image.", "custom_tool", - "Live26 step 6 follow-up -- explicit custom_tool request.", + "Explicit custom_tool request -- writing a tool wrapper, not running an existing one.", ), ] diff --git a/test/evals/datasets/live26_demo.py b/test/evals/datasets/staining_quantification.py similarity index 83% rename from test/evals/datasets/live26_demo.py rename to test/evals/datasets/staining_quantification.py index 33063aeb80f..db2f47e0260 100644 --- a/test/evals/datasets/live26_demo.py +++ b/test/evals/datasets/staining_quantification.py @@ -1,31 +1,33 @@ -"""Live26 (GCC2026) demo dataset: rubric-graded content quality for the demo script. +"""Staining quantification: rubric-graded content quality for an end-to-end +histological staining analysis use case. -These are the prompts we will actually type on stage at GCC2026 for the Live26 -presentation (a histological staining quantification flow ending with an Omero -export). The routing decision for each prompt is scored in -``evals/datasets/routing.py`` under the ``live26_*`` names; this dataset scores -the substance of the response with an LLMJudge per-case rubric, regardless of -which downstream agent the router picks. +The prompts walk the agent through a representative bioimaging flow -- +brightfield RGB inputs, color deconvolution, per-ROI quantification, and +finally exporting results to Omero. They were originally sourced from a +demo script so the casework mirrors what a real user typing into ChatGXY +would do, but the rubrics score domain substance, not stage timing. -Pairs with the planning doc at: -https://docs.google.com/document/d/1-TuXZG-fVRjLDesR3NQFoenBDxJbf7Mt0Sqtokr4DQA +The routing decision for each prompt is scored in +``evals/datasets/routing.py`` under matching case names; this dataset +scores the substance of the response with an LLMJudge per-case rubric, +regardless of which downstream agent the router picks. Notes: -- ``live26_import_iwc_workflow`` depends on the in-flight IWC operations on the +- ``import_iwc_workflow`` depends on the in-flight IWC operations on the ``agent-ops-iwc-reintroduce`` branch (``search_iwc_workflows``, ``import_workflow_from_iwc``). On this branch the case still runs and its rubric just measures how degraded the answer is without those tools -- useful as a "before" number to diff against once IWC ops merges. -- ``live26_history_sanity_check``, ``live26_summarize_to_page``, - ``live26_report_takeaway``, and ``live26_social_media_post`` need a real - Galaxy session because they presuppose specific results in the user's - history. They carry ``requires_galaxy=True`` and are filtered out by - default; pass ``--include-galaxy-required`` to include them. Note that - the harness's MagicMock'd trans means these will still fail until - real-Galaxy plumbing is added; the flag is forward-looking. -- ``live26_report_takeaway`` and ``live26_social_media_post`` don't have a - pinned routing target yet (report-template editing isn't a dedicated agent +- ``history_sanity_check``, ``summarize_to_page``, ``report_takeaway``, + and ``social_media_post`` need a real Galaxy session because they + presuppose specific results in the user's history. They carry + ``requires_galaxy=True`` and are filtered out by default; pass + ``--include-galaxy-required`` to include them. Note that the + standalone CLI's MagicMock'd trans means these will still fail until + the pytest live runner exercises them; the flag is forward-looking. +- ``report_takeaway`` and ``social_media_post`` don't have a pinned + routing target yet (report-template editing isn't a dedicated agent and the social-post beat is borderline); they're content-only here. """ @@ -46,7 +48,7 @@ from pydantic_evals.evaluators import ( _PROTO_CASES: list[dict[str, Any]] = [ { - "name": "live26_stain_quantification_intro", + "name": "stain_quantification_intro", "query": ( "The datasets in my history are brightfield RGB images from a " "histological staining experiment. I'd like to quantify stain " @@ -68,7 +70,7 @@ _PROTO_CASES: list[dict[str, Any]] = [ "requires_galaxy": False, }, { - "name": "live26_import_iwc_workflow", + "name": "import_iwc_workflow", "query": "Import a histological staining workflow from IWC.", "rubric": ( "Response should perform or describe importing an IWC workflow:\n" @@ -85,7 +87,7 @@ _PROTO_CASES: list[dict[str, Any]] = [ "requires_galaxy": False, }, { - "name": "live26_omero_upload_guidance", + "name": "omero_upload_guidance", "query": "How can I upload this data to Omero?", "rubric": ( "Response should guide the user through Omero export from Galaxy:\n" @@ -102,7 +104,7 @@ _PROTO_CASES: list[dict[str, Any]] = [ "requires_galaxy": False, }, { - "name": "live26_history_sanity_check", + "name": "history_sanity_check", "query": "Look at my history -- did I miss anything in this analysis?", "rubric": ( "Response should perform a real sanity check on the user's history:\n" @@ -118,7 +120,7 @@ _PROTO_CASES: list[dict[str, Any]] = [ "requires_galaxy": True, }, { - "name": "live26_summarize_to_page", + "name": "summarize_to_page", "query": "Summarize this analysis and save it as a Galaxy Page.", "rubric": ( "Response should produce a publishable analysis summary:\n" @@ -135,7 +137,7 @@ _PROTO_CASES: list[dict[str, Any]] = [ "requires_galaxy": True, }, { - "name": "live26_custom_tool_quantify_brown", + "name": "custom_tool_quantify_brown", "query": "Generate a Galaxy tool that counts brown pixels in a TIFF image.", "rubric": ( "Response should produce a working Galaxy tool wrapper in the " @@ -158,7 +160,7 @@ _PROTO_CASES: list[dict[str, Any]] = [ "requires_galaxy": False, }, { - "name": "live26_report_takeaway", + "name": "report_takeaway", "query": ( "Add a short take-away message to the workflow report summarizing " "what the staining quantification results show." @@ -178,7 +180,7 @@ _PROTO_CASES: list[dict[str, Any]] = [ "requires_galaxy": True, }, { - "name": "live26_social_media_post", + "name": "social_media_post", "query": "Can you summarize my analysis in a couple of sentences I can share?", "rubric": ( "Response should produce a publishable short summary:\n" @@ -196,8 +198,9 @@ _PROTO_CASES: list[dict[str, Any]] = [ _RUBRIC_TEMPLATE = """\ -You are evaluating a Galaxy AI agent's response to a prompt from the Live26 -(GCC2026) demo script. The demo is a histological staining quantification flow. +You are evaluating a Galaxy AI agent's response to a prompt from the +histological staining quantification use case (brightfield RGB inputs, +color deconvolution, per-ROI quantification, Omero export). Acceptance rubric for this case: {rubric} @@ -212,12 +215,12 @@ Return a number; no commentary. """ -def live26_demo_dataset( +def staining_quantification_dataset( judge_model: Optional[Model] = None, only: Optional[list[str]] = None, include_galaxy_required: bool = False, ) -> Dataset[str, str, dict[str, Any]]: - """Build the live26_demo Dataset. + """Build the staining_quantification Dataset. Requires ``judge_model`` to score; without it the dataset has no evaluators and cases will report no scores. @@ -252,4 +255,4 @@ def live26_demo_dataset( evaluators=evaluators, ) ) - return Dataset(name="live26_demo", cases=cases) + return Dataset(name="staining_quantification", cases=cases) diff --git a/test/evals/seed_live26_demo_history.py b/test/evals/seed_staining_quantification_history.py similarity index 83% rename from test/evals/seed_live26_demo_history.py rename to test/evals/seed_staining_quantification_history.py index 3aec7cb8745..9199db1fc86 100644 --- a/test/evals/seed_live26_demo_history.py +++ b/test/evals/seed_staining_quantification_history.py @@ -1,7 +1,8 @@ -"""Seed a Galaxy history with the mid-state Live26 (GCC2026) demo data. +"""Seed a Galaxy history with mid-state data for the staining quantification +eval use case. -Creates a history named "Live26 staining quantification" populated with the -shape of data the demo flow's later prompts assume: +Creates a history populated with the shape of data the use case's later +prompts assume: - one or more brightfield RGB inputs (stub TIFFs) - a region-of-interest mask @@ -10,23 +11,23 @@ shape of data the demo flow's later prompts assume: The contents are structurally correct but synthetic -- the agents reason about the history's shape, not the pixel values. Swap in real images for -stage rehearsals. +demo rehearsals. Used by: - ``test/integration/test_live_evals.py`` as a fixture for the pytest live-eval runner. -- Stage rehearsal: run standalone against a real Galaxy with an API key - to set up a demo history in seconds. +- Standalone runs against a real Galaxy to set up the demo history in + seconds. Standalone usage (against a running Galaxy): - python test/evals/seed_live26_demo_history.py \\ + python test/evals/seed_staining_quantification_history.py \\ --galaxy-url http://localhost:8080 \\ --galaxy-api-key In-test usage: - from evals.seed_live26_demo_history import seed_demo_history + from evals.seed_staining_quantification_history import seed_demo_history history_id = seed_demo_history(dataset_populator) """ @@ -38,7 +39,7 @@ from typing import ( Optional, ) -HISTORY_NAME = "Live26 staining quantification" +HISTORY_NAME = "Staining quantification (eval fixture)" # Stub TIFF bytes -- structurally correct minimal TIFF header. The agents only # need to see that the dataset exists with the right file type / extension. @@ -56,12 +57,11 @@ _QUANTIFICATION_CSV = """region_id\tarea_pixels\tmean_intensity_brown\tmean_inte """ _HISTORY_ANNOTATION = ( - "Live26 demo: histological staining quantification flow. Brightfield " - "RGB inputs (slide_01, slide_02) -> ROI mask -> color deconvolution " - "isolating the brown stain channel -> per-ROI intensity / area " - "quantification. Final output: staining_quantification_per_roi.tabular " - "with per-region area_pixels, mean_intensity_brown, mean_intensity_blue, " - "and pct_positive." + "Histological staining quantification flow. Brightfield RGB inputs " + "(slide_01, slide_02) -> ROI mask -> color deconvolution isolating " + "the brown stain channel -> per-ROI intensity / area quantification. " + "Final output: staining_quantification_per_roi.tabular with per-region " + "area_pixels, mean_intensity_brown, mean_intensity_blue, and pct_positive." ) diff --git a/test/evals/specs.py b/test/evals/specs.py index 58687b29204..7e30f3d396d 100644 --- a/test/evals/specs.py +++ b/test/evals/specs.py @@ -20,10 +20,10 @@ from galaxy.agents.base import GalaxyAgentDependencies from .datasets import ( bioinformatics_workflows_dataset, error_analysis_dataset, - live26_demo_dataset, orchestrator_planning_dataset, router_tool_use_dataset, routing_dataset, + staining_quantification_dataset, tool_recommendation_dataset, ) from .evaluators import ( @@ -131,14 +131,14 @@ def build_bioinformatics_workflows( ) -def build_live26_demo( +def build_staining_quantification( deps: GalaxyAgentDependencies, judge_model: Optional[Model] = None, only: Optional[list[str]] = None, include_galaxy_required: bool = False, usage_buffer: Optional[list[dict[str, int]]] = None, ) -> BuiltDataset: - dataset = live26_demo_dataset( + dataset = staining_quantification_dataset( judge_model=judge_model, only=only, include_galaxy_required=include_galaxy_required, @@ -173,5 +173,5 @@ SPECS: dict[str, Callable[..., BuiltDataset]] = { "router_tool_use": build_router_tool_use, "bioinformatics_workflows": build_bioinformatics_workflows, "orchestrator_planning": build_orchestrator_planning, - "live26_demo": build_live26_demo, + "staining_quantification": build_staining_quantification, } diff --git a/test/integration/test_live_evals.py b/test/integration/test_live_evals.py index 47b5e4bad11..76c79a511a0 100644 --- a/test/integration/test_live_evals.py +++ b/test/integration/test_live_evals.py @@ -1,15 +1,16 @@ -"""Live-Galaxy runner for the ``evals/`` agent eval suite. +"""Live-Galaxy runner for the ``test/evals/`` agent eval suite. Wraps the standalone ``evals.run_evals`` machinery inside a Galaxy -integration-test fixture so the ``requires_galaxy=True`` cases (live26 -history sanity check, summarize-to-page, report takeaway, social media -post, history_analyzer routing cases) actually run against a real history. +integration-test fixture so the ``requires_galaxy=True`` cases (the +staining quantification history sanity check, summarize-to-page, report +takeaway, social media post, plus the history_analyzer routing cases) +actually run against a real history. -The mock-trans CLI under ``evals/run_evals.py`` stays the fast loop for -prompt iteration on the cases that don't need live data. This test is the -"real flight check" before stage rehearsals -- it shares dataset -definitions, evaluators, and the report renderer with the CLI, only the -deps construction differs. +The mock-trans CLI under ``test/evals/run_evals.py`` stays the fast +loop for prompt iteration on the cases that don't need live data. This +test is the real flight check before demo rehearsals -- it shares +dataset definitions, evaluators, and the report renderer with the CLI, +only the deps construction differs. ## Running @@ -26,13 +27,13 @@ deps construction differs. # Llama-4-Maverick-17B-128E-Instruct # so the candidate isn't judging # its own output - export EVALS_DATASETS="live26_demo" # optional comma-separated subset + export EVALS_DATASETS="staining_quantification" # optional comma-separated subset pytest test/integration/test_live_evals.py -v -Reports land in ``evals/results/--.{md,json}``, the -same place and naming as the CLI so ``--baseline`` diffing keeps working -across both runners. +Reports land in ``test/evals/results/--.{md,json}``, +the same place and naming as the CLI so ``--baseline`` diffing keeps +working across both runners. """ import asyncio @@ -63,7 +64,7 @@ from evals.run_evals import ( # noqa: E402 run_eval_suite, write_eval_report, ) -from evals.seed_live26_demo_history import seed_demo_history # noqa: E402 +from evals.seed_staining_quantification_history import seed_demo_history # noqa: E402 from evals.tasks import make_live_deps # noqa: E402 log = logging.getLogger(__name__) @@ -101,14 +102,17 @@ class TestLiveEvals(IntegrationTestCase): def test_run_live_eval_suite(self): """Seed the demo history, run the eval suite, write reports. - Default scope: ``live26_demo`` dataset with ``--include-galaxy-required`` - so the cases that need a real history actually run. Override via the - ``EVALS_*`` env vars documented at the top of the file. + Default scope: ``staining_quantification`` dataset with + ``--include-galaxy-required`` so the cases that need a real + history actually run. Override via the ``EVALS_*`` env vars + documented at the top of the file. """ history_id = seed_demo_history(self.dataset_populator) - log.info("Seeded Live26 demo history: %s", history_id) + log.info("Seeded staining quantification fixture history: %s", history_id) - datasets = [d.strip() for d in os.environ.get("EVALS_DATASETS", "live26_demo").split(",") if d.strip()] + datasets = [ + d.strip() for d in os.environ.get("EVALS_DATASETS", "staining_quantification").split(",") if d.strip() + ] config_path = os.environ.get("EVALS_MODEL_CONFIG") _path, model_config = _load_model_config(config_path)