From a774a18b95bb10f2afceeea830d008b41e0f99e8 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jos=C3=A9=20Braulio=20Gonz=C3=A1lez=20Valido?= Date: Tue, 14 Jul 2026 11:52:27 +0100 Subject: [PATCH] ci(ai-builder): Fetch eval test cases from LangTracer (no-changelog) (#33982) Co-authored-by: Claude Fable 5 --- .github/WORKFLOWS.md | 9 ++++++--- .github/workflows/ci-mcp-evals.yml | 2 ++ .github/workflows/test-evals-instance-ai.yml | 20 ++++++++++++++++++++ .github/workflows/test-evals-mcp.yml | 19 +++++++++++++++++++ 4 files changed, 47 insertions(+), 3 deletions(-) diff --git a/.github/WORKFLOWS.md b/.github/WORKFLOWS.md index 15eae5c8287..59a02861935 100644 --- a/.github/WORKFLOWS.md +++ b/.github/WORKFLOWS.md @@ -196,7 +196,10 @@ on every push made cost untenable; firing on every review approval cascaded through the dismiss-stale-on-push → re-approve loop, which also blew up. The current trigger fires once per `opened` / `reopened` / `ready_for_review` on a non-fork PR touching the eval surface, and runs the `pr` test-case dataset -(~6 high-reliability, capability-diverse cases) instead of the full ~14. To +(a small set of high-reliability, capability-diverse cases) instead of the full +suite. Test cases are pulled at run time from the LangTracer suite +`n8n-workflows` — the source of truth; CI has no disk fallback (local runs +keep `--source disk` for authoring). To re-run after pushing a fix, dispatch `ci-instance-ai-evals.yml` with the PR number (optionally `tier: full` for broader coverage) — results post back to the PR. The lighter `test-evals-discovery.yml` still runs on every push as part @@ -206,8 +209,8 @@ of `ci-pull-requests.yml`. the lab bench.** The gate deliberately exposes only PR re-runs. Anything that isn't PR gating — baselines, model experiments, arbitrary branch runs — goes through `test-evals-instance-ai.yml`'s own dispatch form ("Instance AI -Evals: Experiments"): full knob set (branch, filter, tier, iterations, experiment-name, -model), no per-PR cancellation (dispatches run in parallel, e.g. concurrent +Evals: Experiments"): full knob set (branch, filter, tier, suite, +iterations, experiment-name, model), no per-PR cancellation (dispatches run in parallel, e.g. concurrent model-comparison arms), and SHA-keyed docker cache hits on master. Evals never run on fork PRs: the event trigger gates on `head.repo.fork`, and the `pr` re-run path refuses fork PRs in `resolve` (dispatched runs carry secrets). diff --git a/.github/workflows/ci-mcp-evals.yml b/.github/workflows/ci-mcp-evals.yml index 211e4e2e17f..3946bbf2c81 100644 --- a/.github/workflows/ci-mcp-evals.yml +++ b/.github/workflows/ci-mcp-evals.yml @@ -69,3 +69,5 @@ jobs: N8N_ENCRYPTION_KEY: ${{ secrets.N8N_ENCRYPTION_KEY }} EVALS_LANGSMITH_ENDPOINT: ${{ secrets.EVALS_LANGSMITH_ENDPOINT }} EVALS_LANGSMITH_API_KEY: ${{ secrets.EVALS_LANGSMITH_API_KEY }} + EVALS_LANGTRACER_URL: ${{ secrets.EVALS_LANGTRACER_URL }} + LANGTRACER_API_KEY: ${{ secrets.LANGTRACER_API_KEY }} diff --git a/.github/workflows/test-evals-instance-ai.yml b/.github/workflows/test-evals-instance-ai.yml index 8d7ce535c0e..3a6def8ba2b 100644 --- a/.github/workflows/test-evals-instance-ai.yml +++ b/.github/workflows/test-evals-instance-ai.yml @@ -18,6 +18,11 @@ on: required: false type: string default: '' + suite: + description: 'LangTracer suite slug or id to pull test cases from.' + required: false + type: string + default: 'n8n-workflows' sandbox-provider: description: 'Sandbox provider (n8n-sandbox or daytona)' required: false @@ -72,6 +77,10 @@ on: description: 'Test-case dataset to run (e.g. "pr", "full"). Empty = no filter.' required: false default: '' + suite: + description: 'LangTracer suite slug or id to pull test cases from.' + required: false + default: 'n8n-workflows' sandbox-provider: description: 'Sandbox provider (n8n-sandbox or daytona)' required: false @@ -345,6 +354,11 @@ jobs: LANGSMITH_PROJECT: instance-ai-evals FILTER: ${{ inputs.filter }} TIER: ${{ inputs.tier }} + SUITE: ${{ inputs.suite }} + # LangTracer is the test-case source of truth (TRUST-247); the CLI + # pulls the suite per run via its export API. + LANGTRACER_URL: ${{ secrets.EVALS_LANGTRACER_URL }} + LANGTRACER_API_KEY: ${{ secrets.LANGTRACER_API_KEY }} ITERATIONS: ${{ inputs.iterations }} EXPERIMENT_NAME: ${{ inputs.experiment-name }} EVAL_PR_NUMBER: ${{ inputs.pr-number || github.event.pull_request.number }} @@ -356,6 +370,12 @@ jobs: done BASE_URLS=$(IFS=,; printf '%s' "${URLS[*]}") ARGS=(--base-url "$BASE_URLS" --concurrency 32 --verbose --iterations "${ITERATIONS:-3}") + # LangTracer is the only CI case source (no disk fallback by design). + ARGS+=(--source langtracer --suite "${SUITE:-n8n-workflows}") + # Pin the LangSmith cohort: langtracer mode otherwise derives a + # suite-scoped dataset/baseline prefix, which would fork the KPI + # history and orphan the baseline comparison. + ARGS+=(--dataset instance-ai-workflow-evals --baseline-prefix instance-ai-baseline-) [ -n "$FILTER" ] && ARGS+=(--filter "$FILTER") [ -n "$TIER" ] && ARGS+=(--tier "$TIER") [ -n "$EXPERIMENT_NAME" ] && ARGS+=(--experiment-name "$EXPERIMENT_NAME") diff --git a/.github/workflows/test-evals-mcp.yml b/.github/workflows/test-evals-mcp.yml index e7f9f66f20a..b449ae95001 100644 --- a/.github/workflows/test-evals-mcp.yml +++ b/.github/workflows/test-evals-mcp.yml @@ -25,6 +25,11 @@ on: required: false type: string default: 'mcp' + suite: + description: 'LangTracer suite slug or id to pull test cases from.' + required: false + type: string + default: 'n8n-workflows' filter: description: 'Only build + eval test cases whose slug matches (comma-separated; empty = whole tier)' required: false @@ -71,6 +76,11 @@ on: required: false EVALS_LANGSMITH_API_KEY: required: false + # The eval CLI pulls the test-case suite from LangTracer. + EVALS_LANGTRACER_URL: + required: false + LANGTRACER_API_KEY: + required: false workflow_dispatch: inputs: branch: @@ -81,6 +91,10 @@ on: description: 'Test-case dataset to build + eval (e.g. "mcp")' required: false default: 'mcp' + suite: + description: 'LangTracer suite slug or id to pull test cases from.' + required: false + default: 'n8n-workflows' filter: description: 'Only build + eval test cases whose slug matches (comma-separated; empty = whole tier)' required: false @@ -254,6 +268,9 @@ jobs: BASE_URLS: ${{ steps.lanes.outputs.base_urls }} LANES: ${{ inputs.lanes || '6' }} TIER: ${{ inputs.tier }} + SUITE: ${{ inputs.suite }} + LANGTRACER_URL: ${{ secrets.EVALS_LANGTRACER_URL }} + LANGTRACER_API_KEY: ${{ secrets.LANGTRACER_API_KEY }} FILTER: ${{ inputs.filter }} ITERATIONS: ${{ inputs.iterations }} EVAL_CONCURRENCY: ${{ inputs.eval-concurrency }} @@ -284,6 +301,8 @@ jobs: --baseline-prefix mcp-baseline- --verbose ) + # LangTracer is the only CI case source (no disk fallback by design). + ARGS+=(--source langtracer --suite "${SUITE:-n8n-workflows}") [ -n "$FILTER" ] && ARGS+=(--filter "$FILTER") [ -n "$EXPERIMENT_NAME" ] && ARGS+=(--experiment-name "$EXPERIMENT_NAME") pnpm eval:instance-ai "${ARGS[@]}"