mirror of
https://github.com/n8n-io/n8n.git
synced 2026-09-24 23:22:38 +08:00
ci(ai-builder): Fetch eval test cases from LangTracer (no-changelog) (#33982)
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
2fc48bdfd5
commit
a774a18b95
@@ -196,7 +196,10 @@ on every push made cost untenable; firing on every review approval cascaded
|
||||
through the dismiss-stale-on-push → re-approve loop, which also blew up.
|
||||
The current trigger fires once per `opened` / `reopened` / `ready_for_review`
|
||||
on a non-fork PR touching the eval surface, and runs the `pr` test-case dataset
|
||||
(~6 high-reliability, capability-diverse cases) instead of the full ~14. To
|
||||
(a small set of high-reliability, capability-diverse cases) instead of the full
|
||||
suite. Test cases are pulled at run time from the LangTracer suite
|
||||
`n8n-workflows` — the source of truth; CI has no disk fallback (local runs
|
||||
keep `--source disk` for authoring). To
|
||||
re-run after pushing a fix, dispatch `ci-instance-ai-evals.yml` with the PR
|
||||
number (optionally `tier: full` for broader coverage) — results post back to
|
||||
the PR. The lighter `test-evals-discovery.yml` still runs on every push as part
|
||||
@@ -206,8 +209,8 @@ of `ci-pull-requests.yml`.
|
||||
the lab bench.** The gate deliberately exposes only PR re-runs. Anything that
|
||||
isn't PR gating — baselines, model experiments, arbitrary branch runs — goes
|
||||
through `test-evals-instance-ai.yml`'s own dispatch form ("Instance AI
|
||||
Evals: Experiments"): full knob set (branch, filter, tier, iterations, experiment-name,
|
||||
model), no per-PR cancellation (dispatches run in parallel, e.g. concurrent
|
||||
Evals: Experiments"): full knob set (branch, filter, tier, suite,
|
||||
iterations, experiment-name, model), no per-PR cancellation (dispatches run in parallel, e.g. concurrent
|
||||
model-comparison arms), and SHA-keyed docker cache hits on master. Evals never
|
||||
run on fork PRs: the event trigger gates on `head.repo.fork`, and the `pr`
|
||||
re-run path refuses fork PRs in `resolve` (dispatched runs carry secrets).
|
||||
|
||||
@@ -69,3 +69,5 @@ jobs:
|
||||
N8N_ENCRYPTION_KEY: ${{ secrets.N8N_ENCRYPTION_KEY }}
|
||||
EVALS_LANGSMITH_ENDPOINT: ${{ secrets.EVALS_LANGSMITH_ENDPOINT }}
|
||||
EVALS_LANGSMITH_API_KEY: ${{ secrets.EVALS_LANGSMITH_API_KEY }}
|
||||
EVALS_LANGTRACER_URL: ${{ secrets.EVALS_LANGTRACER_URL }}
|
||||
LANGTRACER_API_KEY: ${{ secrets.LANGTRACER_API_KEY }}
|
||||
|
||||
@@ -18,6 +18,11 @@ on:
|
||||
required: false
|
||||
type: string
|
||||
default: ''
|
||||
suite:
|
||||
description: 'LangTracer suite slug or id to pull test cases from.'
|
||||
required: false
|
||||
type: string
|
||||
default: 'n8n-workflows'
|
||||
sandbox-provider:
|
||||
description: 'Sandbox provider (n8n-sandbox or daytona)'
|
||||
required: false
|
||||
@@ -72,6 +77,10 @@ on:
|
||||
description: 'Test-case dataset to run (e.g. "pr", "full"). Empty = no filter.'
|
||||
required: false
|
||||
default: ''
|
||||
suite:
|
||||
description: 'LangTracer suite slug or id to pull test cases from.'
|
||||
required: false
|
||||
default: 'n8n-workflows'
|
||||
sandbox-provider:
|
||||
description: 'Sandbox provider (n8n-sandbox or daytona)'
|
||||
required: false
|
||||
@@ -345,6 +354,11 @@ jobs:
|
||||
LANGSMITH_PROJECT: instance-ai-evals
|
||||
FILTER: ${{ inputs.filter }}
|
||||
TIER: ${{ inputs.tier }}
|
||||
SUITE: ${{ inputs.suite }}
|
||||
# LangTracer is the test-case source of truth (TRUST-247); the CLI
|
||||
# pulls the suite per run via its export API.
|
||||
LANGTRACER_URL: ${{ secrets.EVALS_LANGTRACER_URL }}
|
||||
LANGTRACER_API_KEY: ${{ secrets.LANGTRACER_API_KEY }}
|
||||
ITERATIONS: ${{ inputs.iterations }}
|
||||
EXPERIMENT_NAME: ${{ inputs.experiment-name }}
|
||||
EVAL_PR_NUMBER: ${{ inputs.pr-number || github.event.pull_request.number }}
|
||||
@@ -356,6 +370,12 @@ jobs:
|
||||
done
|
||||
BASE_URLS=$(IFS=,; printf '%s' "${URLS[*]}")
|
||||
ARGS=(--base-url "$BASE_URLS" --concurrency 32 --verbose --iterations "${ITERATIONS:-3}")
|
||||
# LangTracer is the only CI case source (no disk fallback by design).
|
||||
ARGS+=(--source langtracer --suite "${SUITE:-n8n-workflows}")
|
||||
# Pin the LangSmith cohort: langtracer mode otherwise derives a
|
||||
# suite-scoped dataset/baseline prefix, which would fork the KPI
|
||||
# history and orphan the baseline comparison.
|
||||
ARGS+=(--dataset instance-ai-workflow-evals --baseline-prefix instance-ai-baseline-)
|
||||
[ -n "$FILTER" ] && ARGS+=(--filter "$FILTER")
|
||||
[ -n "$TIER" ] && ARGS+=(--tier "$TIER")
|
||||
[ -n "$EXPERIMENT_NAME" ] && ARGS+=(--experiment-name "$EXPERIMENT_NAME")
|
||||
|
||||
@@ -25,6 +25,11 @@ on:
|
||||
required: false
|
||||
type: string
|
||||
default: 'mcp'
|
||||
suite:
|
||||
description: 'LangTracer suite slug or id to pull test cases from.'
|
||||
required: false
|
||||
type: string
|
||||
default: 'n8n-workflows'
|
||||
filter:
|
||||
description: 'Only build + eval test cases whose slug matches (comma-separated; empty = whole tier)'
|
||||
required: false
|
||||
@@ -71,6 +76,11 @@ on:
|
||||
required: false
|
||||
EVALS_LANGSMITH_API_KEY:
|
||||
required: false
|
||||
# The eval CLI pulls the test-case suite from LangTracer.
|
||||
EVALS_LANGTRACER_URL:
|
||||
required: false
|
||||
LANGTRACER_API_KEY:
|
||||
required: false
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
branch:
|
||||
@@ -81,6 +91,10 @@ on:
|
||||
description: 'Test-case dataset to build + eval (e.g. "mcp")'
|
||||
required: false
|
||||
default: 'mcp'
|
||||
suite:
|
||||
description: 'LangTracer suite slug or id to pull test cases from.'
|
||||
required: false
|
||||
default: 'n8n-workflows'
|
||||
filter:
|
||||
description: 'Only build + eval test cases whose slug matches (comma-separated; empty = whole tier)'
|
||||
required: false
|
||||
@@ -254,6 +268,9 @@ jobs:
|
||||
BASE_URLS: ${{ steps.lanes.outputs.base_urls }}
|
||||
LANES: ${{ inputs.lanes || '6' }}
|
||||
TIER: ${{ inputs.tier }}
|
||||
SUITE: ${{ inputs.suite }}
|
||||
LANGTRACER_URL: ${{ secrets.EVALS_LANGTRACER_URL }}
|
||||
LANGTRACER_API_KEY: ${{ secrets.LANGTRACER_API_KEY }}
|
||||
FILTER: ${{ inputs.filter }}
|
||||
ITERATIONS: ${{ inputs.iterations }}
|
||||
EVAL_CONCURRENCY: ${{ inputs.eval-concurrency }}
|
||||
@@ -284,6 +301,8 @@ jobs:
|
||||
--baseline-prefix mcp-baseline-
|
||||
--verbose
|
||||
)
|
||||
# LangTracer is the only CI case source (no disk fallback by design).
|
||||
ARGS+=(--source langtracer --suite "${SUITE:-n8n-workflows}")
|
||||
[ -n "$FILTER" ] && ARGS+=(--filter "$FILTER")
|
||||
[ -n "$EXPERIMENT_NAME" ] && ARGS+=(--experiment-name "$EXPERIMENT_NAME")
|
||||
pnpm eval:instance-ai "${ARGS[@]}"
|
||||
|
||||
Reference in New Issue
Block a user