ci(ai-builder): Fetch eval test cases from LangTracer (no-changelog) (#33982)

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
José Braulio González Valido
2026-07-14 10:52:27 +00:00
committed by GitHub
co-authored by Claude Fable 5
parent 2fc48bdfd5
commit a774a18b95
4 changed files with 47 additions and 3 deletions
+6 -3
View File
@@ -196,7 +196,10 @@ on every push made cost untenable; firing on every review approval cascaded
through the dismiss-stale-on-push → re-approve loop, which also blew up.
The current trigger fires once per `opened` / `reopened` / `ready_for_review`
on a non-fork PR touching the eval surface, and runs the `pr` test-case dataset
(~6 high-reliability, capability-diverse cases) instead of the full ~14. To
(a small set of high-reliability, capability-diverse cases) instead of the full
suite. Test cases are pulled at run time from the LangTracer suite
`n8n-workflows` — the source of truth; CI has no disk fallback (local runs
keep `--source disk` for authoring). To
re-run after pushing a fix, dispatch `ci-instance-ai-evals.yml` with the PR
number (optionally `tier: full` for broader coverage) — results post back to
the PR. The lighter `test-evals-discovery.yml` still runs on every push as part
@@ -206,8 +209,8 @@ of `ci-pull-requests.yml`.
the lab bench.** The gate deliberately exposes only PR re-runs. Anything that
isn't PR gating — baselines, model experiments, arbitrary branch runs — goes
through `test-evals-instance-ai.yml`'s own dispatch form ("Instance AI
Evals: Experiments"): full knob set (branch, filter, tier, iterations, experiment-name,
model), no per-PR cancellation (dispatches run in parallel, e.g. concurrent
Evals: Experiments"): full knob set (branch, filter, tier, suite,
iterations, experiment-name, model), no per-PR cancellation (dispatches run in parallel, e.g. concurrent
model-comparison arms), and SHA-keyed docker cache hits on master. Evals never
run on fork PRs: the event trigger gates on `head.repo.fork`, and the `pr`
re-run path refuses fork PRs in `resolve` (dispatched runs carry secrets).
+2
View File
@@ -69,3 +69,5 @@ jobs:
N8N_ENCRYPTION_KEY: ${{ secrets.N8N_ENCRYPTION_KEY }}
EVALS_LANGSMITH_ENDPOINT: ${{ secrets.EVALS_LANGSMITH_ENDPOINT }}
EVALS_LANGSMITH_API_KEY: ${{ secrets.EVALS_LANGSMITH_API_KEY }}
EVALS_LANGTRACER_URL: ${{ secrets.EVALS_LANGTRACER_URL }}
LANGTRACER_API_KEY: ${{ secrets.LANGTRACER_API_KEY }}
@@ -18,6 +18,11 @@ on:
required: false
type: string
default: ''
suite:
description: 'LangTracer suite slug or id to pull test cases from.'
required: false
type: string
default: 'n8n-workflows'
sandbox-provider:
description: 'Sandbox provider (n8n-sandbox or daytona)'
required: false
@@ -72,6 +77,10 @@ on:
description: 'Test-case dataset to run (e.g. "pr", "full"). Empty = no filter.'
required: false
default: ''
suite:
description: 'LangTracer suite slug or id to pull test cases from.'
required: false
default: 'n8n-workflows'
sandbox-provider:
description: 'Sandbox provider (n8n-sandbox or daytona)'
required: false
@@ -345,6 +354,11 @@ jobs:
LANGSMITH_PROJECT: instance-ai-evals
FILTER: ${{ inputs.filter }}
TIER: ${{ inputs.tier }}
SUITE: ${{ inputs.suite }}
# LangTracer is the test-case source of truth (TRUST-247); the CLI
# pulls the suite per run via its export API.
LANGTRACER_URL: ${{ secrets.EVALS_LANGTRACER_URL }}
LANGTRACER_API_KEY: ${{ secrets.LANGTRACER_API_KEY }}
ITERATIONS: ${{ inputs.iterations }}
EXPERIMENT_NAME: ${{ inputs.experiment-name }}
EVAL_PR_NUMBER: ${{ inputs.pr-number || github.event.pull_request.number }}
@@ -356,6 +370,12 @@ jobs:
done
BASE_URLS=$(IFS=,; printf '%s' "${URLS[*]}")
ARGS=(--base-url "$BASE_URLS" --concurrency 32 --verbose --iterations "${ITERATIONS:-3}")
# LangTracer is the only CI case source (no disk fallback by design).
ARGS+=(--source langtracer --suite "${SUITE:-n8n-workflows}")
# Pin the LangSmith cohort: langtracer mode otherwise derives a
# suite-scoped dataset/baseline prefix, which would fork the KPI
# history and orphan the baseline comparison.
ARGS+=(--dataset instance-ai-workflow-evals --baseline-prefix instance-ai-baseline-)
[ -n "$FILTER" ] && ARGS+=(--filter "$FILTER")
[ -n "$TIER" ] && ARGS+=(--tier "$TIER")
[ -n "$EXPERIMENT_NAME" ] && ARGS+=(--experiment-name "$EXPERIMENT_NAME")
+19
View File
@@ -25,6 +25,11 @@ on:
required: false
type: string
default: 'mcp'
suite:
description: 'LangTracer suite slug or id to pull test cases from.'
required: false
type: string
default: 'n8n-workflows'
filter:
description: 'Only build + eval test cases whose slug matches (comma-separated; empty = whole tier)'
required: false
@@ -71,6 +76,11 @@ on:
required: false
EVALS_LANGSMITH_API_KEY:
required: false
# The eval CLI pulls the test-case suite from LangTracer.
EVALS_LANGTRACER_URL:
required: false
LANGTRACER_API_KEY:
required: false
workflow_dispatch:
inputs:
branch:
@@ -81,6 +91,10 @@ on:
description: 'Test-case dataset to build + eval (e.g. "mcp")'
required: false
default: 'mcp'
suite:
description: 'LangTracer suite slug or id to pull test cases from.'
required: false
default: 'n8n-workflows'
filter:
description: 'Only build + eval test cases whose slug matches (comma-separated; empty = whole tier)'
required: false
@@ -254,6 +268,9 @@ jobs:
BASE_URLS: ${{ steps.lanes.outputs.base_urls }}
LANES: ${{ inputs.lanes || '6' }}
TIER: ${{ inputs.tier }}
SUITE: ${{ inputs.suite }}
LANGTRACER_URL: ${{ secrets.EVALS_LANGTRACER_URL }}
LANGTRACER_API_KEY: ${{ secrets.LANGTRACER_API_KEY }}
FILTER: ${{ inputs.filter }}
ITERATIONS: ${{ inputs.iterations }}
EVAL_CONCURRENCY: ${{ inputs.eval-concurrency }}
@@ -284,6 +301,8 @@ jobs:
--baseline-prefix mcp-baseline-
--verbose
)
# LangTracer is the only CI case source (no disk fallback by design).
ARGS+=(--source langtracer --suite "${SUITE:-n8n-workflows}")
[ -n "$FILTER" ] && ARGS+=(--filter "$FILTER")
[ -n "$EXPERIMENT_NAME" ] && ARGS+=(--experiment-name "$EXPERIMENT_NAME")
pnpm eval:instance-ai "${ARGS[@]}"