From a88231544a3cb7550c01601eb593fe8961e680f3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jos=C3=A9=20Braulio=20Gonz=C3=A1lez=20Valido?= Date: Thu, 9 Jul 2026 23:21:46 +0100 Subject: [PATCH] feat(ai-builder): Persist eval expectation verdicts to LangSmith run outputs (no-changelog) (#33788) Co-authored-by: Claude Fable 5 --- .../@n8n/instance-ai/evaluations/README.md | 8 +- .../evaluations/__tests__/aggregator.test.ts | 83 +++++++ .../__tests__/bucket-from-evaluation.test.ts | 124 +++++++++++ .../__tests__/comparison-compare.test.ts | 90 ++++++-- .../__tests__/comparison-format.test.ts | 79 ++++++- .../__tests__/comparison-gate.test.ts | 14 +- .../__tests__/fetch-baseline.test.ts | 200 ++++++++++++++++- .../evaluations/__tests__/reshape.test.ts | 105 ++++++++- .../build-expectations/verifier.ts | 9 +- .../@n8n/instance-ai/evaluations/cli/index.ts | 105 ++++----- .../instance-ai/evaluations/cli/reshape.ts | 50 +++++ .../comparison/bucket-from-evaluation.ts | 77 +++++++ .../evaluations/comparison/compare.ts | 123 +++++++---- .../evaluations/comparison/fetch-baseline.ts | 72 +++++- .../evaluations/comparison/format.ts | 205 +++++++++++------- .../@n8n/instance-ai/evaluations/index.ts | 10 +- .../@n8n/instance-ai/evaluations/types.ts | 6 +- 17 files changed, 1117 insertions(+), 243 deletions(-) create mode 100644 packages/@n8n/instance-ai/evaluations/__tests__/bucket-from-evaluation.test.ts create mode 100644 packages/@n8n/instance-ai/evaluations/comparison/bucket-from-evaluation.ts diff --git a/packages/@n8n/instance-ai/evaluations/README.md b/packages/@n8n/instance-ai/evaluations/README.md index 8544294e3f8..6b18aa27b22 100644 --- a/packages/@n8n/instance-ai/evaluations/README.md +++ b/packages/@n8n/instance-ai/evaluations/README.md @@ -159,7 +159,7 @@ dotenvx run -f ../../../.env.local -- pnpm eval:instance-ai --iterations 3 | `--build-mcp-timeout-ms` | `120000` | `MCP_TIMEOUT` passed to the `claude` build subprocess — bounds one MCP tool call (`--build-via-mcp`) | | `--build-timeout-ms` | `1800000` | Wall-clock cap per build attempt; on expiry the `claude` process is killed so a hung build can't hold its lane. `0` disables. A timed-out build is not retried (`--build-via-mcp`) | -**pass@k / pass^k**: with `--iterations N`, each scenario runs N times. `pass@k` is the fraction of scenarios that passed *at least once*; `pass^k` is the fraction that passed *every* time. `pass@k` shows whether something is *possible*; `pass^k` shows whether it's *reliable*. +**pass@k / pass^k**: with `--iterations N`, each unit — an execution scenario or an evaluated process/outcome expectation — is measured N times. `pass@k` is the fraction of units that passed *at least once*; `pass^k` is the fraction that passed *every* time. `pass@k` shows whether something is *possible*; `pass^k` shows whether it's *reliable*. ### Test-case datasets (logical groupings) @@ -284,7 +284,7 @@ Operational details: ## Regression detection -When `LANGSMITH_API_KEY` is set, every eval run automatically compares its results against the most recent pinned baseline (any experiment whose name starts with `instance-ai-baseline-`). Two output files are written: +When `LANGSMITH_API_KEY` is set, every eval run automatically compares its results against the most recent pinned baseline (any experiment whose name starts with `instance-ai-baseline-`). The comparison is **unit-based**: execution scenarios and evaluated build expectations are compared side by side, keyed `file/scenario` and `file#expectation:text` respectively (`comparison.result.evaluationUnits`, each entry carrying `kind: 'scenario' | 'expectation'`). Expectation verdicts reach the baseline via the per-run `expectationResults` embedded in LangSmith run outputs — a baseline captured before that persistence simply contributes no expectation units, so those show as PR-only (with an explanatory note) until the baseline is refreshed. Renaming an expectation's text drops it out of the intersection until the next refresh — same semantics as renaming a scenario. Two output files are written: - `eval-results.json` — structured data only, including `comparison.result` when a baseline was found. - `eval-pr-comment.md` — the full PR comment rendered as markdown, including the alert, aggregate, comparison sections, per-test-case results, and failure details. Always written; falls back to a no-baseline summary when no comparison ran. @@ -313,9 +313,9 @@ LANGSMITH_API_KEY=... dotenvx run -f ../../../.env.local -- \ LangSmith appends a random suffix (e.g. `instance-ai-baseline-7abc1234`); the most recently started one becomes the comparison target on the next eval run. The comparison is silently skipped on the baseline-creation run itself. -### How scenarios are tiered +### How units are tiered -Each scenario lands in one of three regression tiers, evaluated in order of strictness: +Each unit (scenario or expectation) lands in one of three regression tiers, evaluated in order of strictness: - **Regression** — high-confidence flag, gating-grade. The drop must be statistically significant (chance of seeing it by noise < 5%), at least 30 percentage points in size, and the baseline must have been reliable (≥ 70% pass rate). - **Likely regression** — looser bar for visibility on borderline cases. Looser confidence threshold (chance by noise < 20%), drop ≥ 15 percentage points, baseline ≥ 50%. Frequently natural variance — worth a glance only if your changes touch related code paths. diff --git a/packages/@n8n/instance-ai/evaluations/__tests__/aggregator.test.ts b/packages/@n8n/instance-ai/evaluations/__tests__/aggregator.test.ts index 6565c6c811f..10861b8c9d8 100644 --- a/packages/@n8n/instance-ai/evaluations/__tests__/aggregator.test.ts +++ b/packages/@n8n/instance-ai/evaluations/__tests__/aggregator.test.ts @@ -96,3 +96,86 @@ describe('aggregateResults — verifier-incomplete scenario runs', () => { expect(sa.passAtK).toHaveLength(3); }); }); + +describe('aggregateResults — build expectations as units', () => { + const expectationCase: WorkflowTestCase = { + ...testCase, + executionScenarios: undefined, + processExpectations: ['asks before building'], + outcomeExpectations: ['workflow has a trigger'], + }; + + function expectationRun( + verdicts: Array<{ expectation: string; pass: boolean; incomplete?: boolean }>, + ): WorkflowTestCaseResult { + return { + testCase: expectationCase, + workflowBuildSuccess: true, + executionScenarioResults: [], + buildExpectationResults: verdicts.map((v) => ({ + expectation: v.expectation, + pass: v.pass, + reason: v.pass ? 'ok' : 'nope', + ...(v.incomplete ? { incomplete: true } : {}), + })), + }; + } + + it('aggregates per-expectation counts in process-then-outcome order', () => { + const allRuns = [ + [ + expectationRun([ + { expectation: 'asks before building', pass: true }, + { expectation: 'workflow has a trigger', pass: true }, + ]), + ], + [ + expectationRun([ + { expectation: 'asks before building', pass: true }, + { expectation: 'workflow has a trigger', pass: false }, + ]), + ], + ]; + + const evaluation = aggregateResults(allRuns, 2); + const [process, outcome] = evaluation.testCases[0].buildExpectations; + + expect(process.expectation).toBe('asks before building'); + expect(process).toMatchObject({ evaluatedCount: 2, passCount: 2 }); + expect(process.passAtK).toHaveLength(2); + expect(outcome.expectation).toBe('workflow has a trigger'); + expect(outcome).toMatchObject({ evaluatedCount: 2, passCount: 1 }); + }); + + it('keeps judge-incomplete and missing verdicts out of the denominator', () => { + const allRuns = [ + [ + expectationRun([ + { expectation: 'asks before building', pass: true }, + { expectation: 'workflow has a trigger', pass: true }, + ]), + ], + [ + // Judge returned an incomplete verdict for one expectation and + // nothing at all for the other. + expectationRun([{ expectation: 'asks before building', pass: false, incomplete: true }]), + ], + ]; + + const evaluation = aggregateResults(allRuns, 2); + const [process, outcome] = evaluation.testCases[0].buildExpectations; + + expect(process).toMatchObject({ evaluatedCount: 1, passCount: 1 }); + expect(outcome).toMatchObject({ evaluatedCount: 1, passCount: 1 }); + }); + + it('reports evaluatedCount 0 for an expectation the judge never evaluated', () => { + const allRuns = [[expectationRun([])], [expectationRun([])]]; + + const evaluation = aggregateResults(allRuns, 2); + for (const ea of evaluation.testCases[0].buildExpectations) { + expect(ea).toMatchObject({ evaluatedCount: 0, passCount: 0 }); + expect(ea.passAtK).toEqual([]); + } + }); +}); diff --git a/packages/@n8n/instance-ai/evaluations/__tests__/bucket-from-evaluation.test.ts b/packages/@n8n/instance-ai/evaluations/__tests__/bucket-from-evaluation.test.ts new file mode 100644 index 00000000000..6652c81d240 --- /dev/null +++ b/packages/@n8n/instance-ai/evaluations/__tests__/bucket-from-evaluation.test.ts @@ -0,0 +1,124 @@ +import { bucketFromEvaluation } from '../comparison/bucket-from-evaluation'; +import type { WorkflowTestCaseWithFile } from '../data/workflows'; +import type { + BuildExpectationAggregation, + ExecutionScenarioAggregation, + MultiRunEvaluation, + WorkflowTestCase, +} from '../types'; + +function testCase(): WorkflowTestCase { + return { + conversation: [{ role: 'user', text: 'build it' }], + complexity: 'simple', + tags: [], + datasets: ['full'], + } as WorkflowTestCase; +} + +function scenarioAggregation( + name: string, + runs: Array<{ success: boolean; incomplete?: boolean; failureCategory?: string }>, +): ExecutionScenarioAggregation { + const evaluated = runs.filter((r) => !r.incomplete); + return { + scenario: { name, description: '', dataSetup: '', successCriteria: '' }, + runs: runs.map((r) => ({ + scenario: { name, description: '', dataSetup: '', successCriteria: '' }, + success: r.success, + score: r.success ? 1 : 0, + reasoning: '', + failureCategory: r.failureCategory, + ...(r.incomplete ? { incomplete: true } : {}), + })), + evaluatedCount: evaluated.length, + passCount: evaluated.filter((r) => r.success).length, + passRate: 0, + passAtK: [], + passHatK: [], + }; +} + +function expectationAggregation( + expectation: string, + passCount: number, + evaluatedCount: number, +): BuildExpectationAggregation { + return { + expectation, + runs: [], + evaluatedCount, + passCount, + passRate: evaluatedCount > 0 ? passCount / evaluatedCount : 0, + passAtK: [], + passHatK: [], + }; +} + +function fixture(): { evaluation: MultiRunEvaluation; withFiles: WorkflowTestCaseWithFile[] } { + const tc = testCase(); + const evaluation: MultiRunEvaluation = { + totalRuns: 3, + testCases: [ + { + testCase: tc, + runs: [], + buildSuccessCount: 3, + executionScenarios: [ + scenarioAggregation('happy', [ + { success: true }, + { success: false, failureCategory: 'builder_issue' }, + { success: false, incomplete: true }, + ]), + ], + buildExpectations: [ + expectationAggregation('asks before building', 2, 3), + expectationAggregation('never judged', 0, 0), + ], + }, + ], + }; + return { evaluation, withFiles: [{ testCase: tc, fileSlug: 'my-case' }] }; +} + +describe('bucketFromEvaluation', () => { + it('emits scenario and evaluated-expectation units under their kind-specific keys', () => { + const { evaluation, withFiles } = fixture(); + const bucket = bucketFromEvaluation(evaluation, withFiles, 'pr-exp'); + + expect(bucket.evaluationUnits.get('my-case/happy')).toMatchObject({ + kind: 'scenario', + name: 'happy', + passed: 1, + total: 2, // incomplete run outside the denominator + }); + expect(bucket.evaluationUnits.get('my-case#expectation:asks before building')).toMatchObject({ + kind: 'expectation', + name: 'asks before building', + passed: 2, + total: 3, + }); + }); + + it('excludes expectations with no evaluated verdicts', () => { + const { evaluation, withFiles } = fixture(); + const bucket = bucketFromEvaluation(evaluation, withFiles, 'pr-exp'); + + expect(bucket.evaluationUnits.get('my-case#expectation:never judged')).toBeUndefined(); + expect(bucket.evaluationUnits.size).toBe(2); + }); + + it('keeps trialTotal and failure categories scenario-only', () => { + const { evaluation, withFiles } = fixture(); + const bucket = bucketFromEvaluation(evaluation, withFiles, 'pr-exp'); + + // 2 evaluated scenario runs; expectation trials never counted here. + expect(bucket.trialTotal).toBe(2); + expect(bucket.failureCategoryTotals).toEqual({ builder_issue: 1 }); + }); + + it('throws when a test case has no file slug', () => { + const { evaluation } = fixture(); + expect(() => bucketFromEvaluation(evaluation, [], 'pr-exp')).toThrow(/no fileSlug/); + }); +}); diff --git a/packages/@n8n/instance-ai/evaluations/__tests__/comparison-compare.test.ts b/packages/@n8n/instance-ai/evaluations/__tests__/comparison-compare.test.ts index e6196596df6..66c10dab450 100644 --- a/packages/@n8n/instance-ai/evaluations/__tests__/comparison-compare.test.ts +++ b/packages/@n8n/instance-ai/evaluations/__tests__/comparison-compare.test.ts @@ -1,22 +1,31 @@ import { vi } from 'vitest'; -import { compareBuckets, type ExperimentBucket, type ScenarioCounts } from '../comparison/compare'; +import { + compareBuckets, + unitKeyOf, + type EvaluationUnitCounts, + type ExperimentBucket, +} from '../comparison/compare'; function bucket( name: string, - scenarios: ScenarioCounts[], + units: EvaluationUnitCounts[], categories?: { totals: Record; trialTotal: number }, ): ExperimentBucket { return { experimentName: name, - scenarios: new Map(scenarios.map((s) => [`${s.testCaseFile}/${s.scenarioName}`, s])), + evaluationUnits: new Map(units.map((u) => [unitKeyOf(u), u])), failureCategoryTotals: categories?.totals, trialTotal: categories?.trialTotal, }; } -function s(file: string, scenario: string, passed: number, total: number): ScenarioCounts { - return { testCaseFile: file, scenarioName: scenario, passed, total }; +function s(file: string, scenario: string, passed: number, total: number): EvaluationUnitCounts { + return { kind: 'scenario', testCaseFile: file, name: scenario, passed, total }; +} + +function e(file: string, expectation: string, passed: number, total: number): EvaluationUnitCounts { + return { kind: 'expectation', testCaseFile: file, name: expectation, passed, total }; } describe('compareBuckets', () => { @@ -26,7 +35,7 @@ describe('compareBuckets', () => { const result = compareBuckets(pr, base); - expect(result.scenarios).toHaveLength(2); + expect(result.evaluationUnits).toHaveLength(2); expect(result.prOnly).toEqual([]); expect(result.baselineOnly).toEqual([]); expect(result.aggregate.intersectionSize).toBe(2); @@ -38,12 +47,65 @@ describe('compareBuckets', () => { const result = compareBuckets(pr, base); - expect(result.scenarios).toHaveLength(1); - expect(result.scenarios[0].testCaseFile).toBe('contact'); - expect(result.baselineOnly).toEqual([{ testCaseFile: 'weather', scenarioName: 'happy' }]); + expect(result.evaluationUnits).toHaveLength(1); + expect(result.evaluationUnits[0].testCaseFile).toBe('contact'); + expect(result.baselineOnly).toEqual([ + { kind: 'scenario', testCaseFile: 'weather', name: 'happy' }, + ]); expect(result.prOnly).toEqual([]); }); + it('compares expectation units alongside scenarios and classifies them into tiers', () => { + const pr = bucket('pr', [ + s('contact', 'happy', 10, 10), + e('contact', 'asks before building', 0, 10), + ]); + const base = bucket('master', [ + s('contact', 'happy', 10, 10), + e('contact', 'asks before building', 10, 10), + ]); + + const result = compareBuckets(pr, base); + + expect(result.evaluationUnits).toHaveLength(2); + const expectationUnit = result.evaluationUnits.find((u) => u.kind === 'expectation'); + expect(expectationUnit?.name).toBe('asks before building'); + expect(expectationUnit?.verdict).toBe('hard_regression'); + expect(result.aggregate.intersectionSize).toBe(2); + }); + + it('pools scenario and expectation trials into the aggregate', () => { + const pr = bucket('pr', [s('a', 'happy', 10, 10), e('a', 'no dead ends', 0, 10)]); + const base = bucket('master', [s('a', 'happy', 10, 10), e('a', 'no dead ends', 10, 10)]); + + const result = compareBuckets(pr, base); + + expect(result.aggregate.prAggregatePassRate).toBe(0.5); + expect(result.aggregate.baselineAggregatePassRate).toBe(1); + }); + + it('degrades expectations to prOnly against a baseline captured before expectation persistence', () => { + const pr = bucket('pr', [s('a', 'happy', 9, 10), e('a', 'asks first', 10, 10)]); + const base = bucket('master', [s('a', 'happy', 9, 10)]); // old baseline: scenarios only + + const result = compareBuckets(pr, base); + + expect(result.evaluationUnits).toHaveLength(1); + expect(result.evaluationUnits[0].kind).toBe('scenario'); + expect(result.prOnly).toEqual([{ kind: 'expectation', testCaseFile: 'a', name: 'asks first' }]); + expect(result.aggregate.intersectionSize).toBe(1); + }); + + it('never collides a scenario and an expectation that share a name', () => { + const pr = bucket('pr', [s('a', 'happy', 10, 10), e('a', 'happy', 0, 10)]); + const base = bucket('master', [s('a', 'happy', 10, 10), e('a', 'happy', 10, 10)]); + + const result = compareBuckets(pr, base); + + expect(result.evaluationUnits).toHaveLength(2); + expect(result.evaluationUnits.map((u) => u.kind).sort()).toEqual(['expectation', 'scenario']); + }); + it('aggregates only over the intersection, not over baseline-only or pr-only', () => { const pr = bucket('pr', [s('contact', 'happy', 10, 10)]); const base = bucket('master', [s('contact', 'happy', 5, 10), s('other', 'happy', 0, 10)]); @@ -55,7 +117,7 @@ describe('compareBuckets', () => { expect(result.aggregate.intersectionSize).toBe(1); }); - it('sorts scenarios with regressions first, then improvements, then stable', () => { + it('sorts units with regressions first, then improvements, then stable', () => { const pr = bucket('pr', [ s('a', 'stable', 10, 10), s('b', 'regression', 0, 10), @@ -68,7 +130,7 @@ describe('compareBuckets', () => { ]); const result = compareBuckets(pr, base); - expect(result.scenarios.map((sc) => sc.scenarioName)).toEqual([ + expect(result.evaluationUnits.map((sc) => sc.name)).toEqual([ 'regression', 'improvement', 'stable', @@ -80,7 +142,7 @@ describe('compareBuckets', () => { const base = bucket('master', [s('contact', 'happy', 10, 10)]); const result = compareBuckets(pr, base); - expect(result.scenarios[0].verdict).toBe('insufficient_data'); + expect(result.evaluationUnits[0].verdict).toBe('insufficient_data'); }); it('returns no failure-category drift when either side lacks category totals', () => { @@ -181,12 +243,12 @@ describe('compareBuckets', () => { // Defaults: 5/10 vs 8/10 = -30pp drop, p ≈ 0.18 → soft_regression // (passes soft maxPValue=0.20, soft minDelta=0.15, baseline 80% above soft 50%). const defaults = compareBuckets(pr, base); - expect(defaults.scenarios[0].verdict).toBe('soft_regression'); + expect(defaults.evaluationUnits[0].verdict).toBe('soft_regression'); // Stricter soft p-value cutoff excludes this case. const stricter = compareBuckets(pr, base, { soft: { maxPValue: 0.1, minDelta: 0.15, minBaselinePassRate: 0.5 }, }); - expect(['stable', 'watch']).toContain(stricter.scenarios[0].verdict); + expect(['stable', 'watch']).toContain(stricter.evaluationUnits[0].verdict); }); }); diff --git a/packages/@n8n/instance-ai/evaluations/__tests__/comparison-format.test.ts b/packages/@n8n/instance-ai/evaluations/__tests__/comparison-format.test.ts index 230aabd6742..ae612ea7d91 100644 --- a/packages/@n8n/instance-ai/evaluations/__tests__/comparison-format.test.ts +++ b/packages/@n8n/instance-ai/evaluations/__tests__/comparison-format.test.ts @@ -1,9 +1,10 @@ import { compareBuckets, + unitKeyOf, type ComparisonOutcome, type ComparisonResult, + type EvaluationUnitCounts, type ExperimentBucket, - type ScenarioCounts, } from '../comparison/compare'; import { formatComparisonMarkdown, formatComparisonTerminal } from '../comparison/format'; import type { @@ -21,15 +22,19 @@ function slugMap(evaluation: MultiRunEvaluation, slugs: string[]): Map [tc.testCase, slugs[i] ?? 'unknown'])); } -function bucket(name: string, scenarios: ScenarioCounts[]): ExperimentBucket { +function bucket(name: string, units: EvaluationUnitCounts[]): ExperimentBucket { return { experimentName: name, - scenarios: new Map(scenarios.map((s) => [`${s.testCaseFile}/${s.scenarioName}`, s])), + evaluationUnits: new Map(units.map((u) => [unitKeyOf(u), u])), }; } -function s(file: string, scenario: string, passed: number, total: number): ScenarioCounts { - return { testCaseFile: file, scenarioName: scenario, passed, total }; +function s(file: string, scenario: string, passed: number, total: number): EvaluationUnitCounts { + return { kind: 'scenario', testCaseFile: file, name: scenario, passed, total }; +} + +function e(file: string, expectation: string, passed: number, total: number): EvaluationUnitCounts { + return { kind: 'expectation', testCaseFile: file, name: expectation, passed, total }; } /** Minimal evaluation fixture matching the shape format.ts reads. */ @@ -426,13 +431,13 @@ describe('formatComparisonMarkdown', () => { it('marks new failure categories with 🆕', () => { const pr: ExperimentBucket = { experimentName: 'pr', - scenarios: new Map([['a/happy', { ...s('a', 'happy', 0, 3) }]]), + evaluationUnits: new Map([['a/happy', { ...s('a', 'happy', 0, 3) }]]), failureCategoryTotals: { framework_issue: 9 }, trialTotal: 145, }; const base: ExperimentBucket = { experimentName: 'master', - scenarios: new Map([['a/happy', { ...s('a', 'happy', 5, 10) }]]), + evaluationUnits: new Map([['a/happy', { ...s('a', 'happy', 5, 10) }]]), failureCategoryTotals: { framework_issue: 0 }, trialTotal: 290, }; @@ -651,13 +656,13 @@ describe('formatComparisonMarkdown', () => { // counts on both sides (non-notable but non-zero). const pr: ExperimentBucket = { experimentName: 'pr', - scenarios: new Map([['a/happy', { ...s('a', 'happy', 50, 100) }]]), + evaluationUnits: new Map([['a/happy', { ...s('a', 'happy', 50, 100) }]]), failureCategoryTotals: { builder_issue: 25 }, trialTotal: 100, }; const base: ExperimentBucket = { experimentName: 'master', - scenarios: new Map([['a/happy', { ...s('a', 'happy', 50, 100) }]]), + evaluationUnits: new Map([['a/happy', { ...s('a', 'happy', 50, 100) }]]), failureCategoryTotals: { builder_issue: 22 }, trialTotal: 100, }; @@ -667,6 +672,45 @@ describe('formatComparisonMarkdown', () => { // builder_issue isn't notable here, so no "notable" marker. expect(md).not.toMatch(/builder_issue.*notable/); }); + + it('renders expectation units in the regression tiers with the file :: text label', () => { + const evalWithExpectation = evaluation({ + totalRuns: 3, + testCases: [ + { + userText: 'a', + expectations: [{ text: 'asks before building anything', passes: [false, false, false] }], + }, + ], + }); + const pr = bucket('pr', [e('a', 'asks before building anything', 0, 3)]); + const base = bucket('master', [e('a', 'asks before building anything', 10, 10)]); + const md = formatComparisonMarkdown(evalWithExpectation, ok(compareBuckets(pr, base)), { + slugByTestCase: slugMap(evalWithExpectation, ['a']), + }); + + expect(md).toMatch(/#### Regressions \(1\)/); + expect(md).toMatch(/\| Unit \| PR \| Baseline \| Δ \| p \|/); + expect(md).toContain('`a :: asks before building anything`'); + // The expectation row gets its own failure-breakdown collapsible with judge text. + expect(md).toContain('a :: asks before building anything'); + expect(md).toMatch(/3 of 3 failed/); + }); + + it('labels the with-baseline aggregate with the unit mix and flags expectations missing a baseline', () => { + const pr = bucket('pr', [ + s('a', 'happy', 8, 10), + e('a', 'asks first', 9, 10), + e('a', 'stays quiet', 10, 10), + ]); + const base = bucket('master', [s('a', 'happy', 8, 10), e('a', 'asks first', 9, 10)]); + const md = formatComparisonMarkdown(evalFixture, ok(compareBuckets(pr, base))); + + expect(md).toContain('2 units (1 scenario + 1 expectation)'); + expect(md).toContain( + '1 PR expectations have no baseline data (baseline predates expectation persistence)', + ); + }); }); describe('formatComparisonTerminal', () => { @@ -746,6 +790,21 @@ describe('formatComparisonTerminal', () => { const out = formatComparisonTerminal(agentsEval); - expect(out).toMatch(/Aggregate: 100\.0% pass \(4\/4 trials, 0 scenarios \+ 4 expectations, N=1\)/); + expect(out).toMatch( + /Aggregate: 100\.0% pass \(4\/4 trials, 0 scenarios \+ 4 expectations, N=1\)/, + ); + }); + + it('renders the unit mix in the aggregate heading and expectation rows in tier tables', () => { + const pr = bucket('pr', [s('a', 'happy', 3, 3), e('a', 'asks before building', 0, 3)]); + const base = bucket('master', [ + s('a', 'happy', 10, 10), + e('a', 'asks before building', 10, 10), + ]); + const out = formatComparisonTerminal(evalFixture, ok(compareBuckets(pr, base))); + + expect(out).toMatch(/Aggregate \(2 units \(1 scenario \+ 1 expectation\)\)/); + expect(out).toMatch(/REGRESSIONS/); + expect(out).toContain('a :: asks before building'); }); }); diff --git a/packages/@n8n/instance-ai/evaluations/__tests__/comparison-gate.test.ts b/packages/@n8n/instance-ai/evaluations/__tests__/comparison-gate.test.ts index b884e55e39c..f727dc08d26 100644 --- a/packages/@n8n/instance-ai/evaluations/__tests__/comparison-gate.test.ts +++ b/packages/@n8n/instance-ai/evaluations/__tests__/comparison-gate.test.ts @@ -417,14 +417,20 @@ describe('formatComparisonMarkdown — gate mode', () => { const gate = evaluateGate(evaluation, { slugByTestCase }); const pr = { experimentName: 'pr', - scenarios: new Map([ - ['a/happy', { testCaseFile: 'a', scenarioName: 'happy', passed: 0, total: 3 }], + evaluationUnits: new Map([ + [ + 'a/happy', + { kind: 'scenario' as const, testCaseFile: 'a', name: 'happy', passed: 0, total: 3 }, + ], ]), }; const base = { experimentName: 'master', - scenarios: new Map([ - ['a/happy', { testCaseFile: 'a', scenarioName: 'happy', passed: 10, total: 10 }], + evaluationUnits: new Map([ + [ + 'a/happy', + { kind: 'scenario' as const, testCaseFile: 'a', name: 'happy', passed: 10, total: 10 }, + ], ]), }; const outcome = { kind: 'ok' as const, result: compareBuckets(pr, base) }; diff --git a/packages/@n8n/instance-ai/evaluations/__tests__/fetch-baseline.test.ts b/packages/@n8n/instance-ai/evaluations/__tests__/fetch-baseline.test.ts index ea4d941f5f3..8c90ce61361 100644 --- a/packages/@n8n/instance-ai/evaluations/__tests__/fetch-baseline.test.ts +++ b/packages/@n8n/instance-ai/evaluations/__tests__/fetch-baseline.test.ts @@ -2,7 +2,12 @@ import type { Client } from 'langsmith'; import { vi } from 'vitest'; import type { Mock } from 'vitest'; -import { BASELINE_EXPERIMENT_PREFIX, findLatestBaseline } from '../comparison/fetch-baseline'; +import { + BASELINE_EXPERIMENT_PREFIX, + fetchBaselineBucket, + findLatestBaseline, +} from '../comparison/fetch-baseline'; +import { BUILD_ONLY_SCENARIO_NAME } from '../langsmith/dataset-sync'; interface FakeProject { name?: string; @@ -71,3 +76,196 @@ describe('findLatestBaseline', () => { expect(await findLatestBaseline(client, 'mcp-baseline-')).toBe('mcp-baseline-no-ts'); }); }); + +interface FakeRun { + inputs?: Record; + outputs?: Record | null; +} + +/** Mock a LangSmith client whose `listRuns` yields the given root runs. */ +function bucketClient(runs: FakeRun[]): Client { + return { + readProject: vi.fn(async () => await Promise.resolve({ id: 'proj-1' })), + listRuns: vi.fn(() => + (async function* () { + await Promise.resolve(); + for (const r of runs) yield r; + })(), + ), + } as unknown as Client; +} + +function scenarioRun( + scenarioName: string, + passed: boolean, + extra?: Record, +): FakeRun { + return { + inputs: { testCaseFile: 'my-case', scenarioName }, + outputs: { passed, ...extra }, + }; +} + +describe('fetchBaselineBucket', () => { + it('accumulates per-scenario pass/fail counts across iterations', async () => { + const client = bucketClient([ + scenarioRun('happy-path', true), + scenarioRun('happy-path', false, { failureCategory: 'builder_issue' }), + scenarioRun('edge-case', true), + ]); + const bucket = await fetchBaselineBucket(client, 'instance-ai-baseline-x'); + + expect(bucket.evaluationUnits.get('my-case/happy-path')).toMatchObject({ + kind: 'scenario', + passed: 1, + total: 2, + }); + expect(bucket.evaluationUnits.get('my-case/edge-case')).toMatchObject({ passed: 1, total: 1 }); + expect(bucket.trialTotal).toBe(3); + expect(bucket.failureCategoryTotals).toEqual({ builder_issue: 1 }); + }); + + it('never counts __build_only__ sentinel rows as scenario trials', async () => { + const client = bucketClient([ + scenarioRun(BUILD_ONLY_SCENARIO_NAME, false), + scenarioRun('real-scenario', true), + ]); + const bucket = await fetchBaselineBucket(client, 'instance-ai-baseline-x'); + + expect([...bucket.evaluationUnits.keys()]).toEqual(['my-case/real-scenario']); + expect(bucket.trialTotal).toBe(1); + }); + + it('skips runs with missing/empty outputs and verifier-incomplete rows', async () => { + const client = bucketClient([ + { inputs: { testCaseFile: 'my-case', scenarioName: 'happy-path' }, outputs: null }, + { inputs: { testCaseFile: 'my-case', scenarioName: 'happy-path' }, outputs: {} }, + scenarioRun('happy-path', false, { incomplete: true }), + scenarioRun('happy-path', true), + ]); + const bucket = await fetchBaselineBucket(client, 'instance-ai-baseline-x'); + + expect(bucket.evaluationUnits.get('my-case/happy-path')).toMatchObject({ + passed: 1, + total: 1, + }); + expect(bucket.trialTotal).toBe(1); + }); + + it('ingests expectation verdicts once per (case, iteration) even when every row carries them', async () => { + const verdicts = [ + { expectation: 'asks first', pass: true, reason: 'did' }, + { expectation: 'stays quiet', pass: false, reason: 'did not' }, + ]; + const client = bucketClient([ + // Two scenario rows of the same case+iteration both embed the same verdicts. + { + inputs: { testCaseFile: 'my-case', scenarioName: 's1', _iteration: 0 }, + outputs: { passed: true, expectationResults: verdicts }, + }, + { + inputs: { testCaseFile: 'my-case', scenarioName: 's2', _iteration: 0 }, + outputs: { passed: true, expectationResults: verdicts }, + }, + // Second iteration accumulates on top. + { + inputs: { testCaseFile: 'my-case', scenarioName: 's1', _iteration: 1 }, + outputs: { passed: true, expectationResults: verdicts }, + }, + ]); + const bucket = await fetchBaselineBucket(client, 'instance-ai-baseline-x'); + + expect(bucket.evaluationUnits.get('my-case#expectation:asks first')).toMatchObject({ + kind: 'expectation', + passed: 2, + total: 2, + }); + expect(bucket.evaluationUnits.get('my-case#expectation:stays quiet')).toMatchObject({ + passed: 0, + total: 2, + }); + // Expectation trials never enter the scenario trialTotal. + expect(bucket.trialTotal).toBe(3); + }); + + it('defaults a missing _iteration to 0 so single-iteration rows dedupe together', async () => { + const verdicts = [{ expectation: 'asks first', pass: true, reason: 'did' }]; + const client = bucketClient([ + { + inputs: { testCaseFile: 'my-case', scenarioName: 's1' }, + outputs: { passed: true, expectationResults: verdicts }, + }, + { + inputs: { testCaseFile: 'my-case', scenarioName: 's2', _iteration: 0 }, + outputs: { passed: true, expectationResults: verdicts }, + }, + ]); + const bucket = await fetchBaselineBucket(client, 'instance-ai-baseline-x'); + + expect(bucket.evaluationUnits.get('my-case#expectation:asks first')).toMatchObject({ + passed: 1, + total: 1, + }); + }); + + it('ingests expectations from sentinel and scenario-incomplete rows', async () => { + const client = bucketClient([ + // Build-only case: the sentinel row is the only expectation carrier. + { + inputs: { testCaseFile: 'build-only', scenarioName: BUILD_ONLY_SCENARIO_NAME }, + outputs: { + passed: true, + expectationResults: [{ expectation: 'built it', pass: true, reason: 'ok' }], + }, + }, + // Verifier-incomplete scenario row: skipped as a trial, verdicts still valid. + { + inputs: { testCaseFile: 'my-case', scenarioName: 's1' }, + outputs: { + passed: false, + incomplete: true, + expectationResults: [{ expectation: 'asks first', pass: true, reason: 'ok' }], + }, + }, + ]); + const bucket = await fetchBaselineBucket(client, 'instance-ai-baseline-x'); + + expect(bucket.evaluationUnits.get('build-only#expectation:built it')).toMatchObject({ + passed: 1, + total: 1, + }); + expect(bucket.evaluationUnits.get('my-case#expectation:asks first')).toMatchObject({ + passed: 1, + total: 1, + }); + expect(bucket.trialTotal).toBe(0); + }); + + it('skips judge-incomplete verdicts and produces no expectation units from old baselines', async () => { + const client = bucketClient([ + { + inputs: { testCaseFile: 'my-case', scenarioName: 's1' }, + outputs: { + passed: true, + expectationResults: [ + { expectation: 'asks first', pass: true, reason: 'ok' }, + { expectation: 'no verdict', pass: false, reason: '', incomplete: true }, + ], + }, + }, + // Old-baseline row: no expectationResults field at all. + scenarioRun('legacy-scenario', true), + ]); + const bucket = await fetchBaselineBucket(client, 'instance-ai-baseline-x'); + + expect(bucket.evaluationUnits.get('my-case#expectation:no verdict')).toBeUndefined(); + expect(bucket.evaluationUnits.get('my-case#expectation:asks first')).toMatchObject({ + passed: 1, + total: 1, + }); + const expectationUnits = [...bucket.evaluationUnits.values()].filter( + (u) => u.kind === 'expectation', + ); + expect(expectationUnits).toHaveLength(1); + }); +}); diff --git a/packages/@n8n/instance-ai/evaluations/__tests__/reshape.test.ts b/packages/@n8n/instance-ai/evaluations/__tests__/reshape.test.ts index 0c45ef8a168..fd325831135 100644 --- a/packages/@n8n/instance-ai/evaluations/__tests__/reshape.test.ts +++ b/packages/@n8n/instance-ai/evaluations/__tests__/reshape.test.ts @@ -1,6 +1,10 @@ import type { Run } from 'langsmith/schemas'; -import { reshapeLangSmithRuns } from '../cli/reshape'; +import { + parseTargetOutput, + reshapeLangSmithRuns, + sentinelOutcomeFromVerdicts, +} from '../cli/reshape'; import type { WorkflowTestCaseWithFile } from '../data/workflows'; import { BUILD_ONLY_SCENARIO_NAME } from '../langsmith/dataset-sync'; import type { @@ -317,3 +321,102 @@ describe('reshapeLangSmithRuns', () => { expect(result[0][0]?.runDebug?.[0]?.runId).toBe('run-1'); }); }); + +describe('sentinelOutcomeFromVerdicts', () => { + const pass = (expectation: string): BuildExpectationResult => ({ + expectation, + pass: true, + reason: 'ok', + }); + const fail = (expectation: string): BuildExpectationResult => ({ + expectation, + pass: false, + reason: 'nope', + }); + const noVerdict = (expectation: string): BuildExpectationResult => ({ + expectation, + pass: false, + reason: 'no verdict returned', + incomplete: true, + }); + + it('passes when every evaluated expectation passes', () => { + const out = sentinelOutcomeFromVerdicts([pass('a'), pass('b')]); + expect(out).toMatchObject({ passed: true, score: 1 }); + expect(out.incomplete).toBeUndefined(); + expect(out.failureCategory).toBeUndefined(); + expect(out.reasoning).toContain('all 2 expectations passed'); + }); + + it('fails with a fractional score and names the failed expectations', () => { + const out = sentinelOutcomeFromVerdicts([ + pass('a'), + fail('sends a Slack alert'), + pass('c'), + fail('uses the IF node'), + ]); + expect(out.passed).toBe(false); + expect(out.score).toBeCloseTo(0.5); + expect(out.reasoning).toContain('sends a Slack alert'); + expect(out.reasoning).toContain('uses the IF node'); + }); + + it('excludes incomplete verdicts from the denominator', () => { + const out = sentinelOutcomeFromVerdicts([pass('a'), noVerdict('b')]); + expect(out).toMatchObject({ passed: true, score: 1 }); + expect(out.incomplete).toBeUndefined(); + }); + + it('is incomplete when the judge produced no evaluated verdicts', () => { + for (const verdicts of [undefined, [], [noVerdict('a')]]) { + expect(sentinelOutcomeFromVerdicts(verdicts)).toMatchObject({ + passed: false, + score: 0, + incomplete: true, + }); + } + }); + + // Non-passing sentinels need an explicit category — target() forwards it, and + // without one the feedback extractor labels the LangSmith row 'unknown'. + it('categorizes failed expectations as expectations_failed', () => { + const out = sentinelOutcomeFromVerdicts([pass('a'), fail('b')]); + expect(out.failureCategory).toBe('expectations_failed'); + }); + + it('categorizes judge-dead outcomes as verification_failure', () => { + for (const verdicts of [undefined, [], [noVerdict('a')]]) { + expect(sentinelOutcomeFromVerdicts(verdicts).failureCategory).toBe('verification_failure'); + } + }); +}); + +describe('parseTargetOutput expectationResults', () => { + const base = { buildSuccess: true, passed: true, score: 1, reasoning: 'ok' }; + + it('parses embedded expectation verdicts', () => { + const out = parseTargetOutput({ + ...base, + expectationResults: [ + { expectation: 'a', pass: true, reason: 'did' }, + { expectation: 'b', pass: false, reason: 'no verdict', incomplete: true }, + ], + }); + expect(out?.expectationResults).toEqual([ + { expectation: 'a', pass: true, reason: 'did' }, + { expectation: 'b', pass: false, reason: 'no verdict', incomplete: true }, + ]); + }); + + it('leaves the field undefined when absent', () => { + const out = parseTargetOutput(base); + expect(out).toBeDefined(); + expect(out?.expectationResults).toBeUndefined(); + }); + + it('drops a malformed field without voiding the row', () => { + const out = parseTargetOutput({ ...base, expectationResults: 'garbage' }); + expect(out?.passed).toBe(true); + expect(out?.expectationResults).toBeUndefined(); + }); +}); diff --git a/packages/@n8n/instance-ai/evaluations/build-expectations/verifier.ts b/packages/@n8n/instance-ai/evaluations/build-expectations/verifier.ts index e316a45f253..057906df592 100644 --- a/packages/@n8n/instance-ai/evaluations/build-expectations/verifier.ts +++ b/packages/@n8n/instance-ai/evaluations/build-expectations/verifier.ts @@ -39,10 +39,11 @@ const VERIFY_ATTEMPT_TIMEOUT_MS = 120_000; /** * Judge author-written natural-language expectations about the build conversation + - * resulting workflow. Informational only — never feeds verify_pass@k. On judge failure - * (errors or timeouts across all attempts) it returns `incomplete` verdicts so the report - * stays complete while reading as "no verdict" rather than failures; callers additionally - * guard with `.catch()`. + * resulting workflow. Verdicts are scored as units alongside execution scenarios + * (pass rates, gate) and are embedded in LangSmith run outputs for the baseline + * comparison. On judge failure (errors or timeouts across all attempts) it returns + * `incomplete` verdicts so the report stays complete while reading as "no verdict" + * rather than failures; callers additionally guard with `.catch()`. */ export async function verifyBuildExpectations( expectations: string[], diff --git a/packages/@n8n/instance-ai/evaluations/cli/index.ts b/packages/@n8n/instance-ai/evaluations/cli/index.ts index c787721cd33..839ee17c2f6 100644 --- a/packages/@n8n/instance-ai/evaluations/cli/index.ts +++ b/packages/@n8n/instance-ai/evaluations/cli/index.ts @@ -32,18 +32,18 @@ import { isPlainObject, parseTargetOutput, reshapeLangSmithRuns, + sentinelOutcomeFromVerdicts, type TargetOutput, } from './reshape'; import { aggregateWorkflowChecks, statusMap } from '../binaryChecks/aggregate'; import { selectAuthorExpectations } from '../build-expectations/select'; import { allFailVerdicts, verifyBuildExpectations } from '../build-expectations/verifier'; import { N8nClient } from '../clients/n8n-client'; +import { bucketFromEvaluation } from '../comparison/bucket-from-evaluation'; import { compareBuckets, type ComparisonOutcome, type ComparisonResult, - type ExperimentBucket, - type ScenarioCounts, } from '../comparison/compare'; import { fetchBaselineBucket, findLatestBaseline } from '../comparison/fetch-baseline'; import { @@ -759,7 +759,8 @@ async function runWithLangSmith(config: RunConfig): Promise<{ } // Judge author expectations once per build (off the scenario critical path); - // reshapeLangSmithRuns awaits and merges the verdicts by the build-cache key. + // reshapeLangSmithRuns awaits and merges the verdicts by the build-cache key, + // and target() embeds them in run outputs so baseline fetches can score them. // Full builds judge process + outcome against the real transcript; prebuilt/MCP // builds (no transcript) judge only outcome expectations against the workflow, // with the authored conversation as request context — mirroring the direct loop. @@ -809,8 +810,19 @@ async function runWithLangSmith(config: RunConfig): Promise<{ buildDurationMs, } = await getOrBuild(iteration, inputs.testCaseFile); + // Stashed at build time with a `.catch` attached, so awaiting never rejects. + // Awaited only after each branch's own work is done, keeping the judge off + // the scenario critical path while persisting verdicts to run outputs. + const verdictsPromise = buildExpectationsByKey.get( + `${String(iteration)}:${inputs.testCaseFile}`, + ); + const attachExpectations = async (output: TargetOutput): Promise => { + const verdicts = await verdictsPromise; + return verdicts && verdicts.length > 0 ? { ...output, expectationResults: verdicts } : output; + }; + if (!build.success || !build.workflowId) { - return { + return await attachExpectations({ buildSuccess: false, passed: false, score: 0, @@ -826,19 +838,23 @@ async function runWithLangSmith(config: RunConfig): Promise<{ workflowChecks: build.workflowChecks, buildTrace: build.buildTrace, planRejections: build.proxyDecisionStats?.rejection ?? 0, - }; + }); } - // Build-only case — the build plus its expectation judging (in getOrBuild) is the whole - // test; skip execution. A failed build returns above with its error reasoning; reflect - // the real build status here rather than assuming success. + // Build-only case — the build plus its expectation judging (in getOrBuild) is the + // whole test; skip execution. The sentinel's outcome IS the expectation verdicts, + // so LangSmith pass metrics stay truthful for scenario-less cases. if (inputs.scenarioName === BUILD_ONLY_SCENARIO_NAME) { + const verdicts = await verdictsPromise; + const outcome = sentinelOutcomeFromVerdicts(verdicts); return { buildSuccess: build.success, workflowId: build.workflowId, - passed: false, - score: 0, - reasoning: 'Build-only case — graded by process/outcome expectations', + passed: outcome.passed, + score: outcome.score, + reasoning: outcome.reasoning, + failureCategory: outcome.failureCategory, + ...(outcome.incomplete ? { incomplete: true } : {}), execErrors: [], buildDurationMs, execDurationMs: 0, @@ -847,6 +863,7 @@ async function runWithLangSmith(config: RunConfig): Promise<{ workflowChecks: build.workflowChecks, workflowJson: build.workflowJsons[0], buildTrace: build.buildTrace, + ...(verdicts && verdicts.length > 0 ? { expectationResults: verdicts } : {}), }; } @@ -876,7 +893,7 @@ async function runWithLangSmith(config: RunConfig): Promise<{ // escape to LangSmith, come back as a Run with null outputs, and be // misclassified as builder regressions by the feedback extractor. logger.error(` ERROR [${scenario.name}]: ${errorMessage}`); - return { + return await attachExpectations({ buildSuccess: true, workflowId: build.workflowId, passed: false, @@ -892,7 +909,7 @@ async function runWithLangSmith(config: RunConfig): Promise<{ workflowJson: build.workflowJsons[0], buildTrace: build.buildTrace, planRejections: build.proxyDecisionStats?.rejection ?? 0, - }; + }); } } const execDurationMs = Date.now() - execStart; @@ -902,7 +919,7 @@ async function runWithLangSmith(config: RunConfig): Promise<{ const failureCategory = result.success ? undefined : result.failureCategory; const rootCause = result.success ? undefined : result.rootCause; - return { + return await attachExpectations({ buildSuccess: true, workflowId: build.workflowId, scenarioWorkflowId: result.workflowId, @@ -922,7 +939,7 @@ async function runWithLangSmith(config: RunConfig): Promise<{ workflowJson: build.workflowJsons[0], buildTrace: build.buildTrace, planRejections: build.proxyDecisionStats?.rejection ?? 0, - }; + }); }; const feedbackExtractor = ({ run }: { run: Run }): EvaluationResult[] => { @@ -1241,6 +1258,9 @@ async function writePerRunPassMetrics(config: { if (!exampleId) continue; const output = parseTargetOutput(run.outputs); if (!output) continue; + // Incomplete rows (judge/verifier dead) carry no verdict — keep them out of + // the pass_at_k/pass_hat_k denominator, mirroring feedbackExtractor. + if (output.incomplete) continue; const entry = byExample.get(exampleId) ?? { runIds: [], passed: 0, total: 0 }; entry.runIds.push(run.id); entry.total++; @@ -1611,7 +1631,7 @@ function serializeComparison(result: ComparisonResult): { pr: { experimentName: string }; baseline: { experimentName: string }; aggregate: ComparisonResult['aggregate']; - scenarios: ComparisonResult['scenarios']; + evaluationUnits: ComparisonResult['evaluationUnits']; prOnly: ComparisonResult['prOnly']; baselineOnly: ComparisonResult['baselineOnly']; failureCategories: ComparisonResult['failureCategories']; @@ -1620,7 +1640,7 @@ function serializeComparison(result: ComparisonResult): { pr: result.pr, baseline: result.baseline, aggregate: result.aggregate, - scenarios: result.scenarios, + evaluationUnits: result.evaluationUnits, prOnly: result.prOnly, baselineOnly: result.baselineOnly, failureCategories: result.failureCategories, @@ -1675,57 +1695,6 @@ async function tryRunComparison(config: { } } -/** - * Project the in-memory MultiRunEvaluation onto the bucket shape used by - * fetchBaselineBucket, keyed by `${fileSlug}/${scenarioName}`. - * - * Looks up `fileSlug` by test case reference rather than array index — the - * comparison key depends on getting the right slug, and zipping by index - * silently miscompares if anything ever reorders the aggregate. - */ -function bucketFromEvaluation( - evaluation: MultiRunEvaluation, - testCasesWithFiles: WorkflowTestCaseWithFile[], - experimentName: string, -): ExperimentBucket { - const slugByTestCase = new Map( - testCasesWithFiles.map(({ testCase, fileSlug }) => [testCase, fileSlug]), - ); - const scenarios = new Map(); - const failureCategoryTotals: Record = {}; - let trialTotal = 0; - for (const tc of evaluation.testCases) { - const fileSlug = slugByTestCase.get(tc.testCase); - if (!fileSlug) { - throw new Error( - `bucketFromEvaluation: no fileSlug for test case "${caseDisplayPrompt(tc.testCase, tc.runs[0]?.transcript).slice(0, 60)}"`, - ); - } - for (const sa of tc.executionScenarios) { - const key = `${fileSlug}/${sa.scenario.name}`; - const failureCategories: Record = {}; - for (const sr of sa.runs) { - // Verifier-incomplete runs carry no verdict — not a trial. - if (sr.incomplete) continue; - trialTotal++; - if (!sr.success && sr.failureCategory) { - failureCategories[sr.failureCategory] = (failureCategories[sr.failureCategory] ?? 0) + 1; - failureCategoryTotals[sr.failureCategory] = - (failureCategoryTotals[sr.failureCategory] ?? 0) + 1; - } - } - scenarios.set(key, { - testCaseFile: fileSlug, - scenarioName: sa.scenario.name, - passed: sa.passCount, - total: sa.evaluatedCount, - failureCategories, - }); - } - } - return { experimentName, scenarios, failureCategoryTotals, trialTotal }; -} - main().catch((error) => { console.error('Fatal error:', error); process.exit(1); diff --git a/packages/@n8n/instance-ai/evaluations/cli/reshape.ts b/packages/@n8n/instance-ai/evaluations/cli/reshape.ts index d0c0f30198c..91170028765 100644 --- a/packages/@n8n/instance-ai/evaluations/cli/reshape.ts +++ b/packages/@n8n/instance-ai/evaluations/cli/reshape.ts @@ -32,6 +32,17 @@ const checkOutcomeSchema = z.object({ comment: z.string().optional(), }); +/** Per-expectation verdicts embedded in run outputs — mirrors `BuildExpectationResult` + * so baseline fetches can score expectations alongside scenarios. */ +export const expectationResultsSchema = z.array( + z.object({ + expectation: z.string(), + pass: z.boolean(), + reason: z.string().default(''), + incomplete: z.boolean().optional(), + }), +); + const targetOutputSchema = z.object({ buildSuccess: z.boolean().default(false), passed: z.boolean().default(false), @@ -43,6 +54,8 @@ const targetOutputSchema = z.object({ rootCause: z.string().optional(), /** Verifier returned no verdict — run is excluded from scoring but stays visible. */ incomplete: z.boolean().optional(), + // `.catch` so one malformed field can't void the whole row in `safeParse`. + expectationResults: expectationResultsSchema.optional().catch(undefined), execErrors: z.array(z.string()).default([]), evalResult: z.unknown().optional(), /** Only set on the scenario that initiated the build. */ @@ -123,6 +136,43 @@ export function parseTargetOutput(raw: unknown): TargetOutput | undefined { }; } +/** + * Derive the `__build_only__` sentinel row's outcome from the case's expectation + * verdicts — the judge IS the whole test for a scenario-less case. All evaluated + * expectations passing ⇒ passed; no evaluated verdicts (judge dead) ⇒ `incomplete` + * so the row stays out of scoring instead of reading as a permanent failure. + */ +export function sentinelOutcomeFromVerdicts(verdicts: BuildExpectationResult[] | undefined): { + passed: boolean; + score: number; + reasoning: string; + incomplete?: boolean; + /** Only on non-passing outcomes — without a category the feedback extractor + * files the row under 'unknown' in the LangSmith failure_category column. */ + failureCategory?: 'expectations_failed' | 'verification_failure'; +} { + const evaluated = (verdicts ?? []).filter((v) => !v.incomplete); + if (evaluated.length === 0) { + return { + passed: false, + score: 0, + reasoning: 'Build-only case — no expectation verdicts (judge incomplete)', + incomplete: true, + failureCategory: 'verification_failure', + }; + } + const failed = evaluated.filter((v) => !v.pass); + const passed = failed.length === 0; + return { + passed, + score: (evaluated.length - failed.length) / evaluated.length, + reasoning: passed + ? `Build-only case — all ${String(evaluated.length)} expectations passed` + : `Build-only case — failed expectations: ${failed.map((v) => v.expectation).join('; ')}`, + ...(passed ? {} : { failureCategory: 'expectations_failed' as const }), + }; +} + const runInputsSchema = z .object({ testCaseFile: z.string().default(''), diff --git a/packages/@n8n/instance-ai/evaluations/comparison/bucket-from-evaluation.ts b/packages/@n8n/instance-ai/evaluations/comparison/bucket-from-evaluation.ts new file mode 100644 index 00000000000..f2db10a96e5 --- /dev/null +++ b/packages/@n8n/instance-ai/evaluations/comparison/bucket-from-evaluation.ts @@ -0,0 +1,77 @@ +// --------------------------------------------------------------------------- +// PR-side comparison bucket: project the in-memory MultiRunEvaluation onto +// the ExperimentBucket shape used by fetchBaselineBucket. +// --------------------------------------------------------------------------- + +import { + expectationUnitKey, + scenarioUnitKey, + type EvaluationUnitCounts, + type ExperimentBucket, +} from './compare'; +import type { WorkflowTestCaseWithFile } from '../data/workflows'; +import type { MultiRunEvaluation } from '../types'; +import { caseDisplayPrompt } from '../utils/conversation-text'; + +/** + * Units are execution scenarios plus evaluated build expectations, keyed the + * same way as the baseline bucket. Expectations with no evaluated verdict are + * unmeasured — they don't become units. Failure-category totals and + * `trialTotal` stay scenario-only (expectation verdicts carry no category). + * + * Looks up `fileSlug` by test case reference rather than array index — the + * comparison key depends on getting the right slug, and zipping by index + * silently miscompares if anything ever reorders the aggregate. + */ +export function bucketFromEvaluation( + evaluation: MultiRunEvaluation, + testCasesWithFiles: WorkflowTestCaseWithFile[], + experimentName: string, +): ExperimentBucket { + const slugByTestCase = new Map( + testCasesWithFiles.map(({ testCase, fileSlug }) => [testCase, fileSlug]), + ); + const evaluationUnits = new Map(); + const failureCategoryTotals: Record = {}; + let trialTotal = 0; + for (const tc of evaluation.testCases) { + const fileSlug = slugByTestCase.get(tc.testCase); + if (!fileSlug) { + throw new Error( + `bucketFromEvaluation: no fileSlug for test case "${caseDisplayPrompt(tc.testCase, tc.runs[0]?.transcript).slice(0, 60)}"`, + ); + } + for (const sa of tc.executionScenarios) { + const failureCategories: Record = {}; + for (const sr of sa.runs) { + // Verifier-incomplete runs carry no verdict — not a trial. + if (sr.incomplete) continue; + trialTotal++; + if (!sr.success && sr.failureCategory) { + failureCategories[sr.failureCategory] = (failureCategories[sr.failureCategory] ?? 0) + 1; + failureCategoryTotals[sr.failureCategory] = + (failureCategoryTotals[sr.failureCategory] ?? 0) + 1; + } + } + evaluationUnits.set(scenarioUnitKey(fileSlug, sa.scenario.name), { + kind: 'scenario', + testCaseFile: fileSlug, + name: sa.scenario.name, + passed: sa.passCount, + total: sa.evaluatedCount, + failureCategories, + }); + } + for (const ea of tc.buildExpectations) { + if (ea.evaluatedCount === 0) continue; + evaluationUnits.set(expectationUnitKey(fileSlug, ea.expectation), { + kind: 'expectation', + testCaseFile: fileSlug, + name: ea.expectation, + passed: ea.passCount, + total: ea.evaluatedCount, + }); + } + } + return { experimentName, evaluationUnits, failureCategoryTotals, trialTotal }; +} diff --git a/packages/@n8n/instance-ai/evaluations/comparison/compare.ts b/packages/@n8n/instance-ai/evaluations/comparison/compare.ts index 12bda63913a..545d1e57207 100644 --- a/packages/@n8n/instance-ai/evaluations/comparison/compare.ts +++ b/packages/@n8n/instance-ai/evaluations/comparison/compare.ts @@ -1,6 +1,11 @@ // --------------------------------------------------------------------------- // Comparison core: take two experiment buckets, return a ComparisonResult. // +// A comparable "unit" is an execution scenario or an evaluated build +// expectation — both carry (passed, total) counts and get the same +// statistical treatment. Failure-category drift stays scenario-only: +// expectation verdicts carry no failure category. +// // Pure function, no I/O. The tier thresholds (p-value cutoff, minimum delta, // minimum baseline pass rate) live in statistics.ts — there's no CLI knob. // Tune them there if the false-positive rate drifts. @@ -18,9 +23,13 @@ import { // Types // --------------------------------------------------------------------------- -export interface ScenarioCounts { +export type EvaluationUnitKind = 'scenario' | 'expectation'; + +export interface EvaluationUnitCounts { + kind: EvaluationUnitKind; testCaseFile: string; - scenarioName: string; + /** Scenario name, or the full expectation text for expectation units. */ + name: string; passed: number; total: number; failureCategories?: Record; @@ -28,25 +37,54 @@ export interface ScenarioCounts { export interface ExperimentBucket { experimentName: string; - scenarios: Map; + evaluationUnits: Map; /** - * Aggregated failure-category counts across all trials in all scenarios. + * Aggregated failure-category counts across all *scenario* trials. * Used for the run-level failure-category drift table — orthogonal to - * per-scenario verdicts. + * per-unit verdicts; expectation units never contribute here. */ failureCategoryTotals?: Record; + /** Scenario trials only — the denominator for failure-category rates. */ trialTotal?: number; } -export interface ScenarioComparison extends ScenarioClassification { +/** Bucket key for a scenario unit. */ +export function scenarioUnitKey(testCaseFile: string, scenarioName: string): string { + return `${testCaseFile}/${scenarioName}`; +} + +/** Bucket key for an expectation unit. The `#expectation:` infix cannot occur + * in a `${fileSlug}/${scenarioName}` key, so the two kinds can't collide. */ +export function expectationUnitKey(testCaseFile: string, expectation: string): string { + return `${testCaseFile}#expectation:${expectation}`; +} + +export function unitKeyOf(unit: { + kind: EvaluationUnitKind; testCaseFile: string; - scenarioName: string; + name: string; +}): string { + return unit.kind === 'scenario' + ? scenarioUnitKey(unit.testCaseFile, unit.name) + : expectationUnitKey(unit.testCaseFile, unit.name); +} + +export interface EvaluationUnitComparison extends ScenarioClassification { + kind: EvaluationUnitKind; + testCaseFile: string; + name: string; prPasses: number; prTotal: number; baselinePasses: number; baselineTotal: number; } +export interface UnitRef { + kind: EvaluationUnitKind; + testCaseFile: string; + name: string; +} + export interface AggregateComparison { intersectionSize: number; prAggregatePassRate: number; @@ -70,9 +108,9 @@ export interface ComparisonResult { pr: { experimentName: string }; baseline: { experimentName: string }; aggregate: AggregateComparison; - scenarios: ScenarioComparison[]; - prOnly: Array<{ testCaseFile: string; scenarioName: string }>; - baselineOnly: Array<{ testCaseFile: string; scenarioName: string }>; + evaluationUnits: EvaluationUnitComparison[]; + prOnly: UnitRef[]; + baselineOnly: UnitRef[]; failureCategories: FailureCategoryComparison[]; } @@ -94,22 +132,22 @@ export type ComparisonOutcome = // --------------------------------------------------------------------------- /** Hard regressions only — high-confidence, gating-grade flags. */ -export function hardRegressions(result: ComparisonResult): ScenarioComparison[] { - return result.scenarios.filter((s) => s.verdict === 'hard_regression'); +export function hardRegressions(result: ComparisonResult): EvaluationUnitComparison[] { + return result.evaluationUnits.filter((s) => s.verdict === 'hard_regression'); } /** Soft regressions — looser thresholds, worth investigating but not gating. */ -export function softRegressions(result: ComparisonResult): ScenarioComparison[] { - return result.scenarios.filter((s) => s.verdict === 'soft_regression'); +export function softRegressions(result: ComparisonResult): EvaluationUnitComparison[] { + return result.evaluationUnits.filter((s) => s.verdict === 'soft_regression'); } /** Movement ≥ watchDelta without reaching a flag tier. Visibility only. */ -export function watchList(result: ComparisonResult): ScenarioComparison[] { - return result.scenarios.filter((s) => s.verdict === 'watch'); +export function watchList(result: ComparisonResult): EvaluationUnitComparison[] { + return result.evaluationUnits.filter((s) => s.verdict === 'watch'); } -export function improvements(result: ComparisonResult): ScenarioComparison[] { - return result.scenarios.filter((s) => s.verdict === 'improvement'); +export function improvements(result: ComparisonResult): EvaluationUnitComparison[] { + return result.evaluationUnits.filter((s) => s.verdict === 'improvement'); } export function byVerdict(result: ComparisonResult): Record { @@ -122,7 +160,7 @@ export function byVerdict(result: ComparisonResult): Record = []; - const baselineOnly: Array<{ testCaseFile: string; scenarioName: string }> = []; + const evaluationUnits: EvaluationUnitComparison[] = []; + const prOnly: UnitRef[] = []; + const baselineOnly: UnitRef[] = []; let prIPasses = 0; let prITotal = 0; let baseIPasses = 0; let baseITotal = 0; - for (const [key, prCounts] of pr.scenarios) { - const baseCounts = baseline.scenarios.get(key); + for (const [key, prCounts] of pr.evaluationUnits) { + const baseCounts = baseline.evaluationUnits.get(key); if (!baseCounts) { prOnly.push({ + kind: prCounts.kind, testCaseFile: prCounts.testCaseFile, - scenarioName: prCounts.scenarioName, + name: prCounts.name, }); continue; } @@ -179,9 +219,10 @@ export function compareBuckets( baseCounts.total, options, ); - scenarios.push({ + evaluationUnits.push({ + kind: prCounts.kind, testCaseFile: prCounts.testCaseFile, - scenarioName: prCounts.scenarioName, + name: prCounts.name, prPasses: prCounts.passed, prTotal: prCounts.total, baselinePasses: baseCounts.passed, @@ -190,17 +231,18 @@ export function compareBuckets( }); } - for (const [key, baseCounts] of baseline.scenarios) { - if (!pr.scenarios.has(key)) { + for (const [key, baseCounts] of baseline.evaluationUnits) { + if (!pr.evaluationUnits.has(key)) { baselineOnly.push({ + kind: baseCounts.kind, testCaseFile: baseCounts.testCaseFile, - scenarioName: baseCounts.scenarioName, + name: baseCounts.name, }); } } const aggregate: AggregateComparison = { - intersectionSize: scenarios.length, + intersectionSize: evaluationUnits.length, prAggregatePassRate: rate(prIPasses, prITotal), baselineAggregatePassRate: rate(baseIPasses, baseITotal), prAggregateCI: wilsonInterval(prIPasses, prITotal), @@ -208,7 +250,7 @@ export function compareBuckets( delta: rate(prIPasses, prITotal) - rate(baseIPasses, baseITotal), }; - scenarios.sort(scenarioComparator); + evaluationUnits.sort(unitComparator); const failureCategories = compareFailureCategories(pr, baseline); @@ -216,7 +258,7 @@ export function compareBuckets( pr: { experimentName: pr.experimentName }, baseline: { experimentName: baseline.experimentName }, aggregate, - scenarios, + evaluationUnits, prOnly, baselineOnly, failureCategories, @@ -313,7 +355,7 @@ function rate(passes: number, total: number): number { return total > 0 ? passes / total : 0; } -const VERDICT_ORDER: Record = { +const VERDICT_ORDER: Record = { hard_regression: 0, soft_regression: 1, improvement: 2, @@ -323,11 +365,12 @@ const VERDICT_ORDER: Record = { insufficient_data: 6, }; -function scenarioComparator(a: ScenarioComparison, b: ScenarioComparison): number { +function unitComparator(a: EvaluationUnitComparison, b: EvaluationUnitComparison): number { const av = VERDICT_ORDER[a.verdict]; const bv = VERDICT_ORDER[b.verdict]; if (av !== bv) return av - bv; const fileCmp = a.testCaseFile.localeCompare(b.testCaseFile); if (fileCmp !== 0) return fileCmp; - return a.scenarioName.localeCompare(b.scenarioName); + if (a.kind !== b.kind) return a.kind === 'scenario' ? -1 : 1; + return a.name.localeCompare(b.name); } diff --git a/packages/@n8n/instance-ai/evaluations/comparison/fetch-baseline.ts b/packages/@n8n/instance-ai/evaluations/comparison/fetch-baseline.ts index be5a634f91b..f77cadf2ba0 100644 --- a/packages/@n8n/instance-ai/evaluations/comparison/fetch-baseline.ts +++ b/packages/@n8n/instance-ai/evaluations/comparison/fetch-baseline.ts @@ -13,7 +13,8 @@ // Two functions, both small: // // findLatestBaseline — list baseline-prefixed projects, pick newest. -// fetchBaselineBucket — read its root runs, bucket per scenario. +// fetchBaselineBucket — read its root runs, bucket per evaluation unit +// (execution scenarios + build expectations). // // Both throw on transport errors. Callers are expected to swallow with a log: // the comparison is advisory and shouldn't fail the eval run. @@ -22,7 +23,14 @@ import type { Client } from 'langsmith'; import { z } from 'zod'; -import type { ExperimentBucket, ScenarioCounts } from './compare'; +import { + expectationUnitKey, + scenarioUnitKey, + type EvaluationUnitCounts, + type ExperimentBucket, +} from './compare'; +import { expectationResultsSchema } from '../cli/reshape'; +import { BUILD_ONLY_SCENARIO_NAME } from '../langsmith/dataset-sync'; /** * Prefix the latest-baseline lookup matches against. The CLI flag @@ -37,6 +45,8 @@ const inputsSchema = z .object({ testCaseFile: z.string().default(''), scenarioName: z.string().default(''), + /** 0-based iteration index; absent on single-iteration runs. */ + _iteration: z.number().int().nonnegative().default(0), }) .passthrough(); @@ -45,6 +55,10 @@ const outputsSchema = z passed: z.boolean().default(false), failureCategory: z.string().optional(), incomplete: z.boolean().optional(), + // Case-level expectation verdicts embedded by target(); absent on + // baselines captured before expectation persistence. `.catch` so a + // malformed field doesn't void the whole row. + expectationResults: expectationResultsSchema.optional().catch(undefined), }) .passthrough(); @@ -72,9 +86,16 @@ export async function findLatestBaseline( } /** - * Fetch a baseline experiment's per-scenario pass/fail counts. Each root run + * Fetch a baseline experiment's per-unit pass/fail counts. Each root run * corresponds to one (testCaseFile, scenarioName, iteration) triple — we - * bucket by `${testCaseFile}/${scenarioName}` and accumulate. + * bucket scenarios by `${testCaseFile}/${scenarioName}` and accumulate. + * + * Every row of a case additionally embeds that (case, iteration)'s + * expectation verdicts, so expectations are deduped per (case, iteration) — + * the first row carrying the field wins — and accumulate into expectation + * units. The `__build_only__` sentinel is such a carrier but never a + * scenario unit. Baselines captured before expectation persistence simply + * produce no expectation units. * * Throws if the project does not exist. */ @@ -83,8 +104,9 @@ export async function fetchBaselineBucket( experimentName: string, ): Promise { const project = await client.readProject({ projectName: experimentName }); - const scenarios = new Map(); + const evaluationUnits = new Map(); const failureCategoryTotals: Record = {}; + const seenExpectationCarriers = new Set(); let trialTotal = 0; for await (const run of client.listRuns({ projectId: project.id, isRoot: true })) { @@ -105,13 +127,43 @@ export async function fetchBaselineBucket( } const outputs = outputsSchema.safeParse(rawOutputs); if (!outputs.success) continue; + + // Expectations ingest before the sentinel/incomplete guards: the sentinel + // row is the only carrier for build-only cases, and a scenario-incomplete + // row still holds valid expectation verdicts. + // Same shape as the build-cache key (`iteration:fileSlug`). + const carrierKey = `${String(inputs.data._iteration)}:${inputs.data.testCaseFile}`; + const expectationResults = outputs.data.expectationResults; + if (expectationResults && !seenExpectationCarriers.has(carrierKey)) { + seenExpectationCarriers.add(carrierKey); + for (const verdict of expectationResults) { + // Judge-incomplete verdicts carry no signal — outside the denominator. + if (verdict.incomplete) continue; + const key = expectationUnitKey(inputs.data.testCaseFile, verdict.expectation); + const existing: EvaluationUnitCounts = evaluationUnits.get(key) ?? { + kind: 'expectation', + testCaseFile: inputs.data.testCaseFile, + name: verdict.expectation, + passed: 0, + total: 0, + }; + existing.total++; + if (verdict.pass) existing.passed++; + evaluationUnits.set(key, existing); + } + } + + // Build-only sentinel rows aren't execution scenarios — counting them would + // add a pseudo-scenario per build-only case and skew the trial totals. + if (inputs.data.scenarioName === BUILD_ONLY_SCENARIO_NAME) continue; // Verifier-incomplete rows carry no verdict — skip so they don't count as failed trials. if (outputs.data.incomplete) continue; - const key = `${inputs.data.testCaseFile}/${inputs.data.scenarioName}`; - const existing: ScenarioCounts = scenarios.get(key) ?? { + const key = scenarioUnitKey(inputs.data.testCaseFile, inputs.data.scenarioName); + const existing: EvaluationUnitCounts = evaluationUnits.get(key) ?? { + kind: 'scenario', testCaseFile: inputs.data.testCaseFile, - scenarioName: inputs.data.scenarioName, + name: inputs.data.scenarioName, passed: 0, total: 0, failureCategories: {}, @@ -126,8 +178,8 @@ export async function fetchBaselineBucket( existing.failureCategories[cat] = (existing.failureCategories[cat] ?? 0) + 1; failureCategoryTotals[cat] = (failureCategoryTotals[cat] ?? 0) + 1; } - scenarios.set(key, existing); + evaluationUnits.set(key, existing); } - return { experimentName, scenarios, failureCategoryTotals, trialTotal }; + return { experimentName, evaluationUnits, failureCategoryTotals, trialTotal }; } diff --git a/packages/@n8n/instance-ai/evaluations/comparison/format.ts b/packages/@n8n/instance-ai/evaluations/comparison/format.ts index 5364af548c3..2b29438f5d3 100644 --- a/packages/@n8n/instance-ai/evaluations/comparison/format.ts +++ b/packages/@n8n/instance-ai/evaluations/comparison/format.ts @@ -17,14 +17,18 @@ // --------------------------------------------------------------------------- import { + expectationUnitKey, hardRegressions, improvements, + scenarioUnitKey, softRegressions, + unitKeyOf, watchList, type ComparisonOutcome, type ComparisonResult, + type EvaluationUnitComparison, type FailureCategoryComparison, - type ScenarioComparison, + type UnitRef, } from './compare'; import type { GateCriterion, GateResult, GateUnit } from './gate'; import { aggregateWorkflowChecks } from '../binaryChecks/aggregate'; @@ -124,6 +128,49 @@ function unitCountLabel(summary: { scenarios: number; expectations: number }, to return `${scenarioLabel}${expectationLabel}, N=${totalRuns}`; } +/** Display label for a comparison unit — `file/scenario`, or the + * `file :: expectation-text…` style used by the Failures section. */ +function unitLabel(unit: UnitRef): string { + return unit.kind === 'scenario' + ? `${unit.testCaseFile}/${unit.name}` + : `${unit.testCaseFile} :: ${unit.name.slice(0, 60)}`; +} + +/** `${n} units (X scenarios + Y expectations)` — collapses to the legacy + * `${n} scenarios` copy when no expectation units are present. */ +function unitMixLabel(units: Array<{ kind: EvaluationUnitComparison['kind'] }>): string { + const scenarios = units.filter((u) => u.kind === 'scenario').length; + const expectations = units.length - scenarios; + if (expectations === 0) return `${scenarios} scenario${scenarios === 1 ? '' : 's'}`; + return `${units.length} units (${scenarios} scenario${scenarios === 1 ? '' : 's'} + ${expectations} expectation${expectations === 1 ? '' : 's'})`; +} + +/** Sentence fragments describing units missing on one side of the comparison. + * Expectations missing from the baseline get their own clause — the usual + * cause is a baseline captured before expectation persistence, not case drift. */ +function describePartialCoverage(comparison: ComparisonResult): string[] { + const parts: string[] = []; + if (comparison.baselineOnly.length > 0) { + const label = comparison.baselineOnly.every((u) => u.kind === 'scenario') + ? 'baseline scenarios' + : 'baseline units'; + parts.push(`${comparison.baselineOnly.length} ${label} not run by PR`); + } + const prOnlyScenarios = comparison.prOnly.filter((u) => u.kind === 'scenario').length; + const prOnlyExpectations = comparison.prOnly.length - prOnlyScenarios; + if (prOnlyScenarios > 0) { + parts.push( + `${prOnlyScenarios} PR scenarios have no baseline data (added since baseline captured)`, + ); + } + if (prOnlyExpectations > 0) { + parts.push( + `${prOnlyExpectations} PR expectations have no baseline data (baseline predates expectation persistence)`, + ); + } + return parts; +} + // --------------------------------------------------------------------------- // Markdown PR comment // --------------------------------------------------------------------------- @@ -178,13 +225,11 @@ export function formatComparisonMarkdown( : undefined; if (hard.length > 0) { - lines.push( - ...renderScenarioSection('Regressions', '— high-confidence', hard, true, failedIndex), - ); + lines.push(...renderUnitSection('Regressions', '— high-confidence', hard, true, failedIndex)); } if (soft.length > 0) { lines.push( - ...renderScenarioSection( + ...renderUnitSection( 'Likely regressions', '— looser statistical flag, investigate if related to your changes', soft, @@ -195,7 +240,7 @@ export function formatComparisonMarkdown( } if (watch.length > 0) { lines.push( - ...renderScenarioSection( + ...renderUnitSection( 'Worth watching', '— large change, not flagged as a regression', watch, @@ -205,7 +250,7 @@ export function formatComparisonMarkdown( ); } if (imps.length > 0) { - lines.push(...renderScenarioSection('Improvements', '', imps, true)); + lines.push(...renderUnitSection('Improvements', '', imps, true)); } if (renderedAnyTable) { @@ -544,24 +589,13 @@ function formatAggregateBlock( const arrow = delta > 0 ? ' ↑' : delta < 0 ? ' ↓' : ''; const baselineN = inferBaselineN(comparison); + const mixLabel = unitMixLabel(comparison.evaluationUnits); const sampleLine = baselineN - ? `_${aggregate.intersectionSize} scenarios · N=${evaluation.totalRuns} (PR) vs N=${baselineN} (baseline) · baseline: \`${comparison.baseline.experimentName}\`_` - : `_${aggregate.intersectionSize} scenarios · N=${evaluation.totalRuns} (PR) · baseline: \`${comparison.baseline.experimentName}\`_`; + ? `_${mixLabel} · N=${evaluation.totalRuns} (PR) vs N=${baselineN} (baseline) · baseline: \`${comparison.baseline.experimentName}\`_` + : `_${mixLabel} · N=${evaluation.totalRuns} (PR) · baseline: \`${comparison.baseline.experimentName}\`_`; - const partial = comparison.baselineOnly.length + comparison.prOnly.length; - const partialNote = - partial > 0 - ? `\n_Partial: ${[ - comparison.baselineOnly.length > 0 - ? `${comparison.baselineOnly.length} baseline scenarios not run by PR` - : null, - comparison.prOnly.length > 0 - ? `${comparison.prOnly.length} PR scenarios have no baseline data (added since baseline captured)` - : null, - ] - .filter((s) => s !== null) - .join(', ')}._` - : ''; + const partialParts = describePartialCoverage(comparison); + const partialNote = partialParts.length > 0 ? `\n_Partial: ${partialParts.join(', ')}._` : ''; return [ `**Aggregate**: ${pct(aggregate.prAggregatePassRate)}% PR vs ${pct(aggregate.baselineAggregatePassRate)}% baseline — **${sign}${delta.toFixed(1)}pp${arrow}**`, @@ -569,29 +603,29 @@ function formatAggregateBlock( ].join('\n'); } -function renderScenarioSection( +function renderUnitSection( heading: string, subtitle: string, - scenarios: ScenarioComparison[], + units: EvaluationUnitComparison[], withPValue: boolean, failedIndex?: FailedRunsBySlug, ): string[] { const lines: string[] = []; const headingLine = subtitle - ? `#### ${heading} (${scenarios.length}) ${subtitle}` - : `#### ${heading} (${scenarios.length})`; + ? `#### ${heading} (${units.length}) ${subtitle}` + : `#### ${heading} (${units.length})`; lines.push(headingLine); lines.push(''); if (withPValue) { - lines.push('| Scenario | PR | Baseline | Δ | p |'); + lines.push('| Unit | PR | Baseline | Δ | p |'); lines.push('|---|---|---|---|---|'); } else { - lines.push('| Scenario | PR | Baseline | Δ |'); + lines.push('| Unit | PR | Baseline | Δ |'); lines.push('|---|---|---|---|'); } - for (const s of scenarios) { + for (const s of units) { const cells = [ - `\`${s.testCaseFile}/${s.scenarioName}\``, + `\`${unitLabel(s)}\``, formatRateCell(s.prPasses, s.prTotal), formatRateCell(s.baselinePasses, s.baselineTotal), formatDeltaCell(s.delta), @@ -604,25 +638,25 @@ function renderScenarioSection( } lines.push(''); - // Per-scenario failure breakdown — one collapsible per row that had failed - // PR runs. Lets the reader drill into each flagged scenario without + // Per-unit failure breakdown — one collapsible per row that had failed + // PR runs. Lets the reader drill into each flagged unit without // hunting through a separate "Failure details" section. if (failedIndex) { - for (const s of scenarios) { - const failedRuns = failedIndex.get(`${s.testCaseFile}/${s.scenarioName}`) ?? []; + for (const s of units) { + const failedRuns = failedIndex.get(unitKeyOf(s)) ?? []; if (failedRuns.length === 0) continue; - lines.push(...renderScenarioFailureBreakdown(s, failedRuns)); + lines.push(...renderUnitFailureBreakdown(s, failedRuns)); } } return lines; } -function renderScenarioFailureBreakdown( - s: ScenarioComparison, +function renderUnitFailureBreakdown( + s: EvaluationUnitComparison, failedRuns: FailedRunDetail[], ): string[] { - const slug = `${s.testCaseFile}/${s.scenarioName}`; + const slug = unitLabel(s); const categoryMix = summarizeCategories(failedRuns); const summaryParts = [`${failedRuns.length} of ${s.prTotal} failed`]; if (categoryMix) summaryParts.push(categoryMix); @@ -741,35 +775,31 @@ function renderOtherFindings(comparison: ComparisonResult): string[] { lines.push(`
Other findings: ${summary}`); lines.push(''); - const stableScenarios = comparison.scenarios.filter((s) => s.verdict === 'stable'); - const flakyScenarios = comparison.scenarios.filter((s) => s.verdict === 'unreliable_baseline'); - const noDataScenarios = comparison.scenarios.filter((s) => s.verdict === 'insufficient_data'); + const stableUnits = comparison.evaluationUnits.filter((s) => s.verdict === 'stable'); + const flakyUnits = comparison.evaluationUnits.filter((s) => s.verdict === 'unreliable_baseline'); + const noDataUnits = comparison.evaluationUnits.filter((s) => s.verdict === 'insufficient_data'); - if (flakyScenarios.length > 0) { + if (flakyUnits.length > 0) { lines.push('**Confident drop on a flaky baseline (surfaced for visibility, not flagged):**'); lines.push(''); - lines.push('| Scenario | PR | Baseline | Δ |'); + lines.push('| Unit | PR | Baseline | Δ |'); lines.push('|---|---|---|---|'); - for (const s of flakyScenarios) { + for (const s of flakyUnits) { lines.push( - `| \`${s.testCaseFile}/${s.scenarioName}\` | ${formatRateCell(s.prPasses, s.prTotal)} | ${formatRateCell(s.baselinePasses, s.baselineTotal)} | ${formatDeltaCell(s.delta)} |`, + `| \`${unitLabel(s)}\` | ${formatRateCell(s.prPasses, s.prTotal)} | ${formatRateCell(s.baselinePasses, s.baselineTotal)} | ${formatDeltaCell(s.delta)} |`, ); } lines.push(''); } - if (noDataScenarios.length > 0) { - lines.push( - `**No data:** ${noDataScenarios.map((s) => `\`${s.testCaseFile}/${s.scenarioName}\``).join(', ')}`, - ); + if (noDataUnits.length > 0) { + lines.push(`**No data:** ${noDataUnits.map((s) => `\`${unitLabel(s)}\``).join(', ')}`); lines.push(''); } - if (stableScenarios.length > 0) { - lines.push(`**Stable (${stableScenarios.length}):**`); - lines.push( - stableScenarios.map((s) => `\`${s.testCaseFile}/${s.scenarioName}\``).join(', ') + '.', - ); + if (stableUnits.length > 0) { + lines.push(`**Stable (${stableUnits.length}):**`); + lines.push(stableUnits.map((s) => `\`${unitLabel(s)}\``).join(', ') + '.'); lines.push(''); } @@ -892,7 +922,19 @@ function buildFailedRunsIndex( } }); if (failedRuns.length > 0) { - map.set(`${fileSlug}/${sa.scenario.name}`, failedRuns); + map.set(scenarioUnitKey(fileSlug, sa.scenario.name), failedRuns); + } + } + for (const ea of tc.buildExpectations) { + const failedRuns: FailedRunDetail[] = []; + ea.runs.forEach((r, i) => { + // Judge-incomplete verdicts are outside the comparison denominators — skip. + if (!r.incomplete && !r.pass) { + failedRuns.push({ reasoning: r.reason, runIndex: i + 1 }); + } + }); + if (failedRuns.length > 0) { + map.set(expectationUnitKey(fileSlug, ea.expectation), failedRuns); } } } @@ -933,17 +975,18 @@ function formatDeltaCell(delta: number): string { function countByVerdict( comparison: ComparisonResult, - verdict: ScenarioComparison['verdict'], + verdict: EvaluationUnitComparison['verdict'], ): number { - return comparison.scenarios.filter((s) => s.verdict === verdict).length; + return comparison.evaluationUnits.filter((s) => s.verdict === verdict).length; } -/** Best-effort N=baseline iteration count. The comparison only carries trial - * totals per scenario; we infer N from the most-common scenario total since - * the baseline runs every scenario the same number of times. */ +/** Best-effort N=baseline iteration count, inferred from the most-common + * scenario trial total (the baseline runs every scenario the same number of + * times). Scenario units only — expectation denominators exclude + * judge-incomplete verdicts, so they'd distort the inferred N. */ function inferBaselineN(comparison: ComparisonResult): number | undefined { - const totals = comparison.scenarios - .filter((s) => s.baselineTotal > 0) + const totals = comparison.evaluationUnits + .filter((s) => s.kind === 'scenario' && s.baselineTotal > 0) .map((s) => s.baselineTotal); if (totals.length === 0) return undefined; const counts = new Map(); @@ -1005,7 +1048,7 @@ export function formatComparisonTerminal( TERMINAL_INDENT + 'REGRESSIONS (high-confidence: large drop on a reliable scenario, unlikely noise)', ); - lines.push(formatTerminalScenarioTable(hard, true)); + lines.push(formatTerminalUnitTable(hard, true)); lines.push(''); } if (soft.length > 0) { @@ -1013,17 +1056,17 @@ export function formatComparisonTerminal( TERMINAL_INDENT + 'LIKELY REGRESSIONS (looser statistical flag — investigate if related to your changes)', ); - lines.push(formatTerminalScenarioTable(soft, true)); + lines.push(formatTerminalUnitTable(soft, true)); lines.push(''); } if (watch.length > 0) { lines.push(TERMINAL_INDENT + 'WORTH WATCHING (large change, not flagged as a regression)'); - lines.push(formatTerminalScenarioTable(watch, false)); + lines.push(formatTerminalUnitTable(watch, false)); lines.push(''); } if (imps.length > 0) { lines.push(TERMINAL_INDENT + 'IMPROVEMENTS'); - lines.push(formatTerminalScenarioTable(imps, true)); + lines.push(formatTerminalUnitTable(imps, true)); lines.push(''); } @@ -1115,7 +1158,7 @@ function formatTerminalAggregate( const aggDelta = aggregate.delta * 100; const sign = aggDelta >= 0 ? '+' : ''; const arrow = aggDelta > 0 ? ' ↑' : aggDelta < 0 ? ' ↓' : ''; - lines.push(TERMINAL_INDENT + `Aggregate (${aggregate.intersectionSize} scenarios)`); + lines.push(TERMINAL_INDENT + `Aggregate (${unitMixLabel(comparison.evaluationUnits)})`); lines.push( TERMINAL_INDENT + ` PR ${pct(aggregate.prAggregatePassRate)}% (N=${evaluation.totalRuns})`, @@ -1130,12 +1173,8 @@ function formatTerminalAggregate( } lines.push(TERMINAL_INDENT + ` Δ ${sign}${aggDelta.toFixed(1)}pp${arrow}`); - if (comparison.baselineOnly.length > 0 || comparison.prOnly.length > 0) { - const partialParts: string[] = []; - if (comparison.baselineOnly.length > 0) - partialParts.push(`${comparison.baselineOnly.length} baseline scenarios not run by PR`); - if (comparison.prOnly.length > 0) - partialParts.push(`${comparison.prOnly.length} PR scenarios have no baseline data`); + const partialParts = describePartialCoverage(comparison); + if (partialParts.length > 0) { lines.push(TERMINAL_INDENT + ` partial: ${partialParts.join(', ')}`); } @@ -1254,28 +1293,28 @@ function formatTerminalPerTestCase( return lines; } -function formatTerminalScenarioTable(scenarios: ScenarioComparison[], withPValue: boolean): string { - const names = scenarios.map((s) => `${s.testCaseFile}/${s.scenarioName}`); - const prCells = scenarios.map((s) => `${s.prPasses}/${s.prTotal}`); - const baseCells = scenarios.map((s) => `${s.baselinePasses}/${s.baselineTotal}`); - const deltaCells = scenarios.map((s) => { +function formatTerminalUnitTable(units: EvaluationUnitComparison[], withPValue: boolean): string { + const names = units.map((s) => unitLabel(s)); + const prCells = units.map((s) => `${s.prPasses}/${s.prTotal}`); + const baseCells = units.map((s) => `${s.baselinePasses}/${s.baselineTotal}`); + const deltaCells = units.map((s) => { const d = s.delta * 100; const sign = d >= 0 ? '+' : ''; const arrow = d > 0 ? ' ↑' : d < 0 ? ' ↓' : ''; return `${sign}${d.toFixed(0)}pp${arrow}`; }); const pCells = withPValue - ? scenarios.map((s) => (s.verdict === 'improvement' ? s.pValueRight : s.pValueLeft).toFixed(3)) + ? units.map((s) => (s.verdict === 'improvement' ? s.pValueRight : s.pValueLeft).toFixed(3)) : []; - const nameW = maxWidth(names, 'scenario'); + const nameW = maxWidth(names, 'unit'); const prW = maxWidth(prCells, 'PR'); const baseW = maxWidth(baseCells, 'baseline'); const deltaW = maxWidth(deltaCells, 'Δ'); const pW = withPValue ? maxWidth(pCells, 'p') : 0; const headers = [ - 'scenario'.padEnd(nameW), + 'unit'.padEnd(nameW), 'PR'.padEnd(prW), 'baseline'.padEnd(baseW), 'Δ'.padEnd(deltaW), @@ -1284,7 +1323,7 @@ function formatTerminalScenarioTable(scenarios: ScenarioComparison[], withPValue const widths = withPValue ? [nameW, prW, baseW, deltaW, pW] : [nameW, prW, baseW, deltaW]; const sep = widths.map((w) => '─'.repeat(w)).join(' '); - const rows = scenarios.map((_, i) => { + const rows = units.map((_, i) => { const cells = [ names[i].padEnd(nameW), prCells[i].padEnd(prW), diff --git a/packages/@n8n/instance-ai/evaluations/index.ts b/packages/@n8n/instance-ai/evaluations/index.ts index e7123aa46a7..a684eab6deb 100644 --- a/packages/@n8n/instance-ai/evaluations/index.ts +++ b/packages/@n8n/instance-ai/evaluations/index.ts @@ -52,15 +52,21 @@ export { hardRegressions, softRegressions, watchList, + scenarioUnitKey, + expectationUnitKey, + unitKeyOf, } from './comparison/compare'; export type { ComparisonResult, - ScenarioComparison, - ScenarioCounts, + EvaluationUnitComparison, + EvaluationUnitCounts, + EvaluationUnitKind, + UnitRef, ExperimentBucket, AggregateComparison, FailureCategoryComparison, } from './comparison/compare'; +export { bucketFromEvaluation } from './comparison/bucket-from-evaluation'; export { classifyScenario, fishersExactOneSidedLeft, diff --git a/packages/@n8n/instance-ai/evaluations/types.ts b/packages/@n8n/instance-ai/evaluations/types.ts index 87fcbdefa62..f52bf493af4 100644 --- a/packages/@n8n/instance-ai/evaluations/types.ts +++ b/packages/@n8n/instance-ai/evaluations/types.ts @@ -249,7 +249,8 @@ export interface ExecutionScenarioResult { incomplete?: boolean; } -/** Verdict for one author-written build expectation. Informational only. */ +/** Verdict for one author-written build expectation. Scored as a unit in the + * pass rate alongside execution scenarios. */ export interface BuildExpectationResult { expectation: string; pass: boolean; @@ -274,7 +275,8 @@ export interface WorkflowTestCaseResult { workflowChecks?: CheckOutcome[]; /** Captured build-time sub-agent/tool activity for builder debugging. */ buildTrace?: BuildTrace; - /** Per-expectation verdicts from the build-expectations judge. Not consumed by pass@k. */ + /** Per-expectation verdicts from the build-expectations judge. Aggregated as + * scoring units alongside execution scenarios. */ buildExpectationResults?: BuildExpectationResult[]; /** Base URL of the n8n instance behind this run. Per-result so multi-lane * configs each get their own URL for canvas/execution links. */