From dde77cb7d39943ef6e6ae8e468c5ab772bdadef8 Mon Sep 17 00:00:00 2001 From: Arvin A <51036481+DeveloperTheExplorer@users.noreply.github.com> Date: Fri, 17 Jul 2026 11:46:40 +0200 Subject: [PATCH] feat(editor): Add per-case compare tabs to eval collection compare view (#33880) --- .../@n8n/api-types/src/frontend-settings.ts | 7 + packages/@n8n/api-types/src/index.ts | 3 + .../__tests__/eval-collections.schema.test.ts | 38 +++ .../src/schemas/eval-collections.schema.ts | 39 +++ .../evaluation-collection.service.test.ts | 42 +++ .../evaluation-collection.service.ts | 20 +- .../__tests__/eval-insights.service.test.ts | 30 +++ .../insights/eval-insights.service.ts | 59 +++-- .../__tests__/test-runner.service.ee.test.ts | 43 +++ .../test-runner/test-runner.service.ee.ts | 29 ++- packages/cli/src/services/frontend.service.ts | 1 + .../frontend/@n8n/i18n/src/locales/en.json | 32 +++ .../editor-ui/src/__tests__/defaults.ts | 1 + .../components/Compare/CasesTable.test.ts | 123 +++++++++ .../components/Compare/CasesTable.vue | 246 ++++++++++++++++++ .../components/Compare/CompareHeader.test.ts | 16 ++ .../components/Compare/CompareHeader.vue | 30 ++- .../components/Compare/CompareTabs.test.ts | 92 +++++++ .../components/Compare/CompareTabs.vue | 127 +++++++++ .../Compare/DatasetMismatchBanner.test.ts | 19 ++ .../Compare/DatasetMismatchBanner.vue | 25 ++ .../components/Compare/MetricCriteria.vue | 97 +++++++ .../components/Compare/MetricsTab.vue | 102 ++++++++ .../components/Compare/OutputsTab.test.ts | 79 ++++++ .../components/Compare/OutputsTab.vue | 220 ++++++++++++++++ .../components/Compare/ScoreChart.vue | 4 + .../Compare/WorkflowDiffTab.test.ts | 74 ++++++ .../components/Compare/WorkflowDiffTab.vue | 227 ++++++++++++++++ .../CollectionCard.vue | 35 ++- .../UngroupedRunRow.vue | 17 +- .../components/ListRuns/RunsSection.vue | 63 ++++- .../shared/GroupedMetricChart.test.ts | 2 +- .../components/shared/GroupedMetricChart.vue | 4 +- .../composables/useCompareCases.test.ts | 193 ++++++++++++++ .../composables/useCompareCases.ts | 195 ++++++++++++++ .../composables/useCompareData.test.ts | 16 +- .../composables/useCompareData.ts | 16 +- .../useEvalCollectionsFlag.test.ts | 41 +++ .../composables/useEvalCollectionsFlag.ts | 28 +- .../ai/evaluation.ee/evaluation.utils.test.ts | 23 ++ .../ai/evaluation.ee/evaluation.utils.ts | 130 ++++++--- .../views/CompareCollectionView.test.ts | 44 ++++ .../views/CompareCollectionView.vue | 96 ++++++- .../views/EvalCollectionsListView.vue | 211 +++++++++------ .../views/EvaluationsListSwitcher.vue | 33 ++- .../evaluation.ee/views/EvaluationsView.vue | 4 + .../playwright/pages/EvaluationComparePage.ts | 62 +++++ packages/testing/playwright/pages/n8nPage.ts | 3 + .../e2e/ai/eval-collections-compare.spec.ts | 209 +++++++++++++++ 49 files changed, 3014 insertions(+), 236 deletions(-) create mode 100644 packages/@n8n/api-types/src/schemas/__tests__/eval-collections.schema.test.ts create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CasesTable.test.ts create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CasesTable.vue create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareTabs.test.ts create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareTabs.vue create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/DatasetMismatchBanner.test.ts create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/DatasetMismatchBanner.vue create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/MetricCriteria.vue create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/MetricsTab.vue create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/OutputsTab.test.ts create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/OutputsTab.vue create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/WorkflowDiffTab.test.ts create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/WorkflowDiffTab.vue create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareCases.test.ts create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareCases.ts create mode 100644 packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useEvalCollectionsFlag.test.ts create mode 100644 packages/testing/playwright/pages/EvaluationComparePage.ts create mode 100644 packages/testing/playwright/tests/e2e/ai/eval-collections-compare.spec.ts diff --git a/packages/@n8n/api-types/src/frontend-settings.ts b/packages/@n8n/api-types/src/frontend-settings.ts index 304db7d9f78..7595a5b2a9d 100644 --- a/packages/@n8n/api-types/src/frontend-settings.ts +++ b/packages/@n8n/api-types/src/frontend-settings.ts @@ -257,6 +257,13 @@ export interface FrontendSettings { easyAIWorkflowOnboarded: boolean; evaluation: { quota: number; + /** + * Operator override (`N8N_EVAL_COLLECTIONS_ENABLED`) that force-enables the + * eval-collections surface. Surfaced here so the frontend gate works even + * when the in-browser PostHog client is disabled (telemetry off), where the + * `084_eval_collections` flag would otherwise never resolve. + */ + collectionsEnabled: boolean; }; /** Backend modules that were initialized during startup. */ diff --git a/packages/@n8n/api-types/src/index.ts b/packages/@n8n/api-types/src/index.ts index e26a0bec78e..1ec8d86d970 100644 --- a/packages/@n8n/api-types/src/index.ts +++ b/packages/@n8n/api-types/src/index.ts @@ -515,6 +515,9 @@ export { export { EVAL_COLLECTIONS_FLAG, + RESERVED_METRIC_KEYS, + ONE_TO_FIVE_METRIC_KEYS, + normalizeMetricScore, evalCollectionVersionEntrySchema, createEvaluationCollectionSchema, CreateEvaluationCollectionDto, diff --git a/packages/@n8n/api-types/src/schemas/__tests__/eval-collections.schema.test.ts b/packages/@n8n/api-types/src/schemas/__tests__/eval-collections.schema.test.ts new file mode 100644 index 00000000000..367b0bfb6c2 --- /dev/null +++ b/packages/@n8n/api-types/src/schemas/__tests__/eval-collections.schema.test.ts @@ -0,0 +1,38 @@ +import { normalizeMetricScore, RESERVED_METRIC_KEYS } from '../eval-collections.schema'; + +describe('normalizeMetricScore', () => { + it('excludes reserved operational metrics', () => { + for (const key of RESERVED_METRIC_KEYS) { + // Even a value that happens to land in [0, 1] is not a score. + expect(normalizeMetricScore(key, 0.5)).toBeNull(); + expect(normalizeMetricScore(key, 1719)).toBeNull(); + } + }); + + it('normalizes 1–5 AI-judge metrics onto [0, 1]', () => { + expect(normalizeMetricScore('correctness', 5)).toBe(1); + expect(normalizeMetricScore('correctness', 4)).toBe(0.8); + expect(normalizeMetricScore('correctness', 1)).toBe(0.2); + expect(normalizeMetricScore('helpfulness', 2.5)).toBe(0.5); + }); + + it('drops AI-judge values that fall outside the 1–5 range once scaled', () => { + // 6 / 5 = 1.2 → out of [0, 1] + expect(normalizeMetricScore('correctness', 6)).toBeNull(); + expect(normalizeMetricScore('helpfulness', -1)).toBeNull(); + }); + + it('passes through other metrics only when already in [0, 1]', () => { + expect(normalizeMetricScore('accuracy', 0.9)).toBe(0.9); + expect(normalizeMetricScore('accuracy', 0)).toBe(0); + expect(normalizeMetricScore('accuracy', 1)).toBe(1); + // Unknown-scale values outside [0, 1] can't be scaled → excluded. + expect(normalizeMetricScore('accuracy', 1.5)).toBeNull(); + expect(normalizeMetricScore('accuracy', -0.2)).toBeNull(); + }); + + it('returns null for NaN', () => { + expect(normalizeMetricScore('accuracy', Number.NaN)).toBeNull(); + expect(normalizeMetricScore('correctness', Number.NaN)).toBeNull(); + }); +}); diff --git a/packages/@n8n/api-types/src/schemas/eval-collections.schema.ts b/packages/@n8n/api-types/src/schemas/eval-collections.schema.ts index 9c0dfabb24f..fdbc3afb65d 100644 --- a/packages/@n8n/api-types/src/schemas/eval-collections.schema.ts +++ b/packages/@n8n/api-types/src/schemas/eval-collections.schema.ts @@ -14,6 +14,45 @@ export type EvalCollectionRunStatus = 'new' | 'running' | 'completed' | 'error' // available after `083_canvas_nodes_grouping`). export const EVAL_COLLECTIONS_FLAG = '084_eval_collections'; +// Metric keys every run emits automatically (token counts + execution time). +// They're absolute operational values, not quality scores, so the "avg score" +// derivation and the score-shaped charts exclude them and count only +// user-defined metrics. Single-sourced so FE and BE agree on what a "score" +// is (the FE builds its `PREDEFINED_METRIC_KEYS` set from this). +export const RESERVED_METRIC_KEYS = [ + 'promptTokens', + 'completionTokens', + 'totalTokens', + 'executionTime', +] as const; + +// User-defined metrics whose LLM-as-judge handlers return a 1–5 rating (rather +// than a 0–1 fraction). Every other user metric is assumed already normalized +// to [0, 1]. Single-sourced so FE and BE agree; the FE's `getMetricCategory` +// derives its `aiBased` category from this list. +export const ONE_TO_FIVE_METRIC_KEYS = ['correctness', 'helpfulness'] as const; + +/** + * Normalize a raw metric value to a [0, 1] "score", or return null when the + * metric isn't a score we can chart/average: + * - reserved operational metrics (token counts, execution time) → null + * - 1–5 AI-judge metrics → value / 5 + * - any other metric → passed through only if already in [0, 1]; an + * unknown-scale value outside that range can't be meaningfully scaled, so + * it's excluded rather than rendered as a bogus percentage. + * + * Shared by the FE compare surfaces and the BE avg-score/insights derivation + * so a "score" means the same thing everywhere. + */ +export function normalizeMetricScore(key: string, value: number): number | null { + if ((RESERVED_METRIC_KEYS as readonly string[]).includes(key)) return null; + if (Number.isNaN(value)) return null; + const normalized = (ONE_TO_FIVE_METRIC_KEYS as readonly string[]).includes(key) + ? value / 5 + : value; + return normalized >= 0 && normalized <= 1 ? normalized : null; +} + // Per-version entry on a create-collection request. Either reference an // existing test run (`existingTestRunId`) to reuse it, or omit it and the // service will schedule a fresh run pinned to `workflowVersionId`. A null diff --git a/packages/cli/src/evaluation.ee/__tests__/evaluation-collection.service.test.ts b/packages/cli/src/evaluation.ee/__tests__/evaluation-collection.service.test.ts index 3dc8d295907..4f3de103026 100644 --- a/packages/cli/src/evaluation.ee/__tests__/evaluation-collection.service.test.ts +++ b/packages/cli/src/evaluation.ee/__tests__/evaluation-collection.service.test.ts @@ -452,6 +452,48 @@ describe('EvaluationCollectionService', () => { expect(b?.lastRun?.isCritical).toBe(true); }); + it('scores versions on normalized quality metrics, excluding token/time metrics', async () => { + const versions: WorkflowHistory[] = [ + mock({ + versionId: 'wfv-a', + name: 'A', + autosaved: false, + createdAt: new Date('2026-04-01'), + }), + mock({ + versionId: 'wfv-b', + name: 'B', + autosaved: false, + createdAt: new Date('2026-04-02'), + }), + ]; + workflowHistoryRepo.find.mockResolvedValueOnce(versions); + // correctness is a 1–5 metric → /5; tokens/executionTime are operational + // and excluded, so avgScore is the correctness score alone rather than a + // token-dominated mean in the thousands. + testRunRepo.find.mockResolvedValueOnce([ + makeTestRun({ + id: 'tr-a', + workflowVersionId: 'wfv-a', + metrics: { correctness: 5, completionTokens: 1719, executionTime: 21368 }, + }), + makeTestRun({ + id: 'tr-b', + workflowVersionId: 'wfv-b', + metrics: { correctness: 2, completionTokens: 540, executionTime: 11646 }, + }), + ]); + + const result = await service.getEvalVersions('wf-1', 'cfg-1'); + + const a = result.versions.find((v) => v.workflowVersionId === 'wfv-a'); + const b = result.versions.find((v) => v.workflowVersionId === 'wfv-b'); + expect(a?.lastRun?.avgScore).toBe(1); // 5 / 5 + expect(b?.lastRun?.avgScore).toBeCloseTo(0.4); // 2 / 5 + expect(a?.lastRun?.isBest).toBe(true); + expect(b?.lastRun?.isCritical).toBe(true); // 0.4 < 0.6 + }); + it('includes a current-draft row with no last run', async () => { workflowHistoryRepo.find.mockResolvedValueOnce([]); const result = await service.getEvalVersions('wf-1', 'cfg-1'); diff --git a/packages/cli/src/evaluation.ee/evaluation-collection.service.ts b/packages/cli/src/evaluation.ee/evaluation-collection.service.ts index 8cc90d0d524..95b59b2cd9d 100644 --- a/packages/cli/src/evaluation.ee/evaluation-collection.service.ts +++ b/packages/cli/src/evaluation.ee/evaluation-collection.service.ts @@ -7,6 +7,7 @@ import type { EvaluationCollectionRunSummary, UpdateEvaluationCollectionPayload, } from '@n8n/api-types'; +import { normalizeMetricScore } from '@n8n/api-types'; import type { TestRun, User } from '@n8n/db'; import { EvaluationCollectionRepository, @@ -179,6 +180,11 @@ export class EvaluationCollectionService { workflowVersionId: versionId, evaluationConfigId: input.evaluationConfigId, evaluationConfigSnapshot: configSnapshot, + // Compile the eval config (dataset + trigger + metric nodes) onto + // each version's snapshot. Without this the run goes "direct" and + // the raw versioned workflow has no evaluation trigger → the run + // fails immediately with EVALUATION_TRIGGER_NOT_FOUND. + compileFromConfig: true, }, ); runsStartedIds.push(testRun.id); @@ -474,10 +480,16 @@ export class EvaluationCollectionService { private computeAvgScore(run: TestRun): number | null { const coerced = this.coerceMetrics(run.metrics); if (!coerced) return null; - const values = Object.values(coerced); - if (values.length === 0) return null; - const sum = values.reduce((acc, v) => acc + v, 0); - return sum / values.length; + // A "score" is a user-defined metric normalized to [0, 1] by its scale + // (AI-judge metrics are 1–5 → /5). Operational metrics (token counts, + // execution time) normalize to null and are excluded. Mirrors the FE's + // score model so the versions table's %/best/critical annotations match + // the compare view; null when a run reports no score metric. + const scores = Object.entries(coerced) + .map(([key, value]) => normalizeMetricScore(key, value)) + .filter((value): value is number => value !== null); + if (scores.length === 0) return null; + return scores.reduce((acc, value) => acc + value, 0) / scores.length; } private coerceMetrics( diff --git a/packages/cli/src/evaluation.ee/insights/__tests__/eval-insights.service.test.ts b/packages/cli/src/evaluation.ee/insights/__tests__/eval-insights.service.test.ts index 6684a69d03c..0d44a4877a0 100644 --- a/packages/cli/src/evaluation.ee/insights/__tests__/eval-insights.service.test.ts +++ b/packages/cli/src/evaluation.ee/insights/__tests__/eval-insights.service.test.ts @@ -218,6 +218,36 @@ describe('EvalInsightsService', () => { expect(fluencyRegression).toBeUndefined(); }); + it('scores on normalized quality metrics, ignoring token/time operational metrics', async () => { + collectionRepo.getDetailByIdAndWorkflowId.mockResolvedValueOnce({ + collection: makeCollection(), + runs: [ + // A wins on correctness (5/5) despite far more tokens/time; B is + // worse on correctness (2/5). The winner + the only regression must + // be about correctness — never completionTokens / executionTime. + makeRun({ + id: 'tr-a', + metrics: { correctness: 5, completionTokens: 1719, executionTime: 21368 }, + }), + makeRun({ + id: 'tr-b', + metrics: { correctness: 2, completionTokens: 540, executionTime: 11646 }, + }), + ], + }); + + const result = await service.generateInsights(user, 'wf-1', 'col-1'); + + expect(result.insights.winner.versionLabel).toBe('A'); + // correctness 5/5 → 100%, counted as a single score metric + expect(result.insights.winner.body).toContain('100%'); + expect(result.insights.winner.body).toContain('1 metric'); + const regressedMetrics = result.insights.regressions.map((r) => r.metric); + expect(regressedMetrics).toContain('correctness'); + expect(regressedMetrics).not.toContain('completionTokens'); + expect(regressedMetrics).not.toContain('executionTime'); + }); + it('returns a "lock in baseline" recommendation when no regressions cross the threshold', async () => { collectionRepo.getDetailByIdAndWorkflowId.mockResolvedValueOnce({ collection: makeCollection(), diff --git a/packages/cli/src/evaluation.ee/insights/eval-insights.service.ts b/packages/cli/src/evaluation.ee/insights/eval-insights.service.ts index 09119001e9e..1c15cc71bc1 100644 --- a/packages/cli/src/evaluation.ee/insights/eval-insights.service.ts +++ b/packages/cli/src/evaluation.ee/insights/eval-insights.service.ts @@ -1,5 +1,5 @@ import type { AiInsightsPayload, AiInsightsResponse } from '@n8n/api-types'; -import { aiInsightsResponseSchema } from '@n8n/api-types'; +import { aiInsightsResponseSchema, normalizeMetricScore } from '@n8n/api-types'; import { LicenseState, Logger } from '@n8n/backend-common'; import type { TestRun, User } from '@n8n/db'; import { EvaluationCollectionRepository } from '@n8n/db'; @@ -27,7 +27,10 @@ type RunSummary = { versionLabel: string; workflowVersionId: string | null; avgScore: number | null; - metrics: Record; + // Per-metric scores normalized to [0, 1] (operational metrics excluded). + // Winner/regression comparisons run over these, not the raw metrics, so the + // insights talk about actual quality scores rather than token totals. + scores: Record; }; /** @@ -156,33 +159,31 @@ export class EvalInsightsService { // ---- internals ---- private summariseRun(run: TestRun, index: number): RunSummary { - const metrics = this.coerceMetrics(run.metrics); - const avg = this.averageScore(metrics); + const scores = this.scoreMetrics(run.metrics); + const values = Object.values(scores); + const avgScore = + values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : null; // Label by index letter — A/B/C — so the FE legend chips line up. The // agent (and the deterministic path) refer back to these labels. const versionLabel = String.fromCharCode(0x41 + index); - return { - versionLabel, - workflowVersionId: run.workflowVersionId, - avgScore: avg, - metrics, - }; + return { versionLabel, workflowVersionId: run.workflowVersionId, avgScore, scores }; } - private averageScore(metrics: Record): number | null { - const values = Object.values(metrics); - if (values.length === 0) return null; - return values.reduce((sum, v) => sum + v, 0) / values.length; - } - - private coerceMetrics( + // Per-metric scores normalized to [0, 1] by their scale (AI-judge metrics + // are 1–5 → /5); operational metrics (token counts, execution time) and + // unknown-scale values are dropped. Booleans coerce to 0/1. Shares the + // `normalizeMetricScore` contract with the FE + versions-table scoring so + // winner/regressions reflect the same "score" the user sees. + private scoreMetrics( metrics: Record | null | undefined, ): Record { if (!metrics) return {}; const out: Record = {}; - for (const [k, v] of Object.entries(metrics)) { - if (typeof v === 'number') out[k] = v; - else if (typeof v === 'boolean') out[k] = v ? 1 : 0; + for (const [key, raw] of Object.entries(metrics)) { + const value = typeof raw === 'boolean' ? (raw ? 1 : 0) : raw; + if (typeof value !== 'number') continue; + const score = normalizeMetricScore(key, value); + if (score !== null) out[key] = score; } return out; } @@ -252,7 +253,7 @@ export class EvalInsightsService { winner: { versionLabel: winner.versionLabel, headline: `${winner.versionLabel} is the winner`, - body: `${winner.versionLabel} leads on average score (${this.formatScore(winner.avgScore)}) across ${Object.keys(winner.metrics).length} metric(s).`, + body: `${winner.versionLabel} leads on average score (${this.formatScore(winner.avgScore)}) across ${Object.keys(winner.scores).length} metric(s).`, }, regressions, suggestedNext, @@ -269,17 +270,20 @@ export class EvalInsightsService { const regressions: AiInsightsPayload['regressions'] = []; for (const run of scored) { if (run.versionLabel === winner.versionLabel) continue; - for (const [metric, winnerScore] of Object.entries(winner.metrics)) { - const runScore = run.metrics[metric]; + for (const [metric, winnerScore] of Object.entries(winner.scores)) { + const runScore = run.scores[metric]; if (typeof runScore !== 'number') continue; const delta = runScore - winnerScore; if (delta >= -REGRESSION_DELTA_THRESHOLD) continue; + // `delta` is a [-1, 1] score difference; report it as signed + // percentage points so the copy reads "N percentage points below". + const deltaPoints = Number((delta * 100).toFixed(1)); regressions.push({ versionLabel: run.versionLabel, metric, - delta: Number((delta * 100).toFixed(1)), + delta: deltaPoints, headline: `${run.versionLabel} regressed on ${metric}`, - body: `${run.versionLabel} scored ${this.formatScore(runScore)} on ${metric}, ${this.formatScore(Math.abs(delta * 100))} percentage points below ${winner.versionLabel}.`, + body: `${run.versionLabel} scored ${this.formatScore(runScore)} on ${metric}, ${Math.abs(deltaPoints).toFixed(1)} percentage points below ${winner.versionLabel}.`, }); } } @@ -306,7 +310,8 @@ export class EvalInsightsService { } private formatScore(score: number): string { - // Compact 0–1 score formatting; agent prose can override per locale. - return score.toFixed(2); + // Scores are normalized to [0, 1]; surface them as whole-percent so the + // prose reads "83%" rather than "0.83". Agent prose can override per locale. + return `${Math.round(score * 100)}%`; } } diff --git a/packages/cli/src/evaluation.ee/test-runner/__tests__/test-runner.service.ee.test.ts b/packages/cli/src/evaluation.ee/test-runner/__tests__/test-runner.service.ee.test.ts index adb5d815f45..e6e395a423b 100644 --- a/packages/cli/src/evaluation.ee/test-runner/__tests__/test-runner.service.ee.test.ts +++ b/packages/cli/src/evaluation.ee/test-runner/__tests__/test-runner.service.ee.test.ts @@ -2706,6 +2706,49 @@ describe('TestRunnerService', () => { ); }); + test('compiles from the supplied snapshot and does not re-read the live config', async () => { + // Collection runs pass a frozen config snapshot so a config edit racing + // the run can't change later versions' compilation. The runner must + // compile from that snapshot and skip the live DB lookup. + evaluationConfigRepository.findByIdAndWorkflowId.mockClear(); + workflowRepository.findById.mockResolvedValueOnce({ + id: 'wf-1', + name: 'Live', + nodes: [{ name: 'LiveNode' }], + connections: {}, + settings: {}, + } as never); + workflowHistoryService.findVersion.mockResolvedValueOnce({ + versionId: 'wfv-x', + nodes: [{ name: 'SnapshotNode' } as never], + connections: {} as never, + } as never); + testRunRepository.createTestRun.mockResolvedValueOnce(mock({ id: 'tr-snap' })); + + const snapshot = { id: 'cfg-1', name: 'frozen', metrics: [] } as never; + let compiledConfig: unknown; + workflowCompiler.compile.mockImplementationOnce((wf, config) => { + compiledConfig = config; + return wf as never; + }); + + const { finished } = await testRunnerService.startTestRun(USER as never, 'wf-1', 1, { + collectionId: 'col-1', + workflowVersionId: 'wfv-x', + evaluationConfigId: 'cfg-1', + evaluationConfigSnapshot: snapshot, + compileFromConfig: true, + }); + await finished.catch(() => undefined); + + expect(evaluationConfigRepository.findByIdAndWorkflowId).not.toHaveBeenCalled(); + expect(compiledConfig).toBe(snapshot); + expect(testRunRepository.createTestRun).toHaveBeenCalledWith( + 'wf-1', + expect.objectContaining({ evaluationConfigSnapshot: snapshot }), + ); + }); + test('overwrites workflowData.versionId with the pinned history versionId', async () => { // `ExecutionPersistence` reads `workflowData.versionId` and stores // it on the execution row. If the live workflow's `versionId` leaks diff --git a/packages/cli/src/evaluation.ee/test-runner/test-runner.service.ee.ts b/packages/cli/src/evaluation.ee/test-runner/test-runner.service.ee.ts index 68709eeb880..a40e95727d9 100644 --- a/packages/cli/src/evaluation.ee/test-runner/test-runner.service.ee.ts +++ b/packages/cli/src/evaluation.ee/test-runner/test-runner.service.ee.ts @@ -620,16 +620,25 @@ export class TestRunnerService { let evaluationConfigSnapshot = options?.evaluationConfigSnapshot ?? null; let configToCompile: EvaluationConfig | undefined; let configLookupErrorCode: typeof TestRunErrorCode.EVALUATION_CONFIG_NOT_FOUND | undefined; - if (options?.compileFromConfig && options?.evaluationConfigId) { - const config = await this.evaluationConfigRepository.findByIdAndWorkflowId( - options.evaluationConfigId, - workflowId, - ); - if (!config) { - configLookupErrorCode = TestRunErrorCode.EVALUATION_CONFIG_NOT_FOUND; - } else { - configToCompile = config; - evaluationConfigSnapshot = config as unknown as IDataObject; + if (options?.compileFromConfig) { + if (options.evaluationConfigSnapshot) { + // Prefer the caller-supplied frozen snapshot: collection runs pass one + // so every version compiles against identical dataset/trigger/metrics, + // immune to a config edit that races the run. Its serialized dates are + // not read by the compiler, which only needs the eval fields. + configToCompile = options.evaluationConfigSnapshot as unknown as EvaluationConfig; + } else if (options.evaluationConfigId) { + // No snapshot supplied (single-run callers) — resolve the live config. + const config = await this.evaluationConfigRepository.findByIdAndWorkflowId( + options.evaluationConfigId, + workflowId, + ); + if (!config) { + configLookupErrorCode = TestRunErrorCode.EVALUATION_CONFIG_NOT_FOUND; + } else { + configToCompile = config; + evaluationConfigSnapshot = config as unknown as IDataObject; + } } } diff --git a/packages/cli/src/services/frontend.service.ts b/packages/cli/src/services/frontend.service.ts index 44a671654f4..4368e11fff7 100644 --- a/packages/cli/src/services/frontend.service.ts +++ b/packages/cli/src/services/frontend.service.ts @@ -419,6 +419,7 @@ export class FrontendService { }, evaluation: { quota: this.licenseState.getMaxWorkflowsWithEvaluations(), + collectionsEnabled: this.globalConfig.evaluation.collectionsEnabled, }, activeModules: this.moduleRegistry.getActiveModules(), canvasOnly: this.globalConfig.canvasOnly, diff --git a/packages/frontend/@n8n/i18n/src/locales/en.json b/packages/frontend/@n8n/i18n/src/locales/en.json index c0882c6ad0e..0da7db0e2e7 100644 --- a/packages/frontend/@n8n/i18n/src/locales/en.json +++ b/packages/frontend/@n8n/i18n/src/locales/en.json @@ -5721,6 +5721,7 @@ "evaluation.collections.errors.fetchFailed": "Could not load evaluation collections", "evaluation.collections.card.done": "Done", "evaluation.collections.card.running": "Running", + "evaluation.collections.card.failed": "Failed", "evaluation.collections.card.currentDraft": "Current draft", "evaluation.collections.card.meta.versions": "{count} version | {count} versions", "evaluation.collections.card.lastRunToday": "today, {time}", @@ -5756,6 +5757,37 @@ "evaluation.compare.insights.noRegressions": "No regressions detected", "evaluation.compare.errors.loadFailed": "Could not load the comparison.", "evaluation.compare.errors.notFound": "This collection could not be found.", + "evaluation.compare.tabs.cases": "Cases", + "evaluation.compare.tabs.outputs": "Outputs", + "evaluation.compare.tabs.metrics": "Metrics", + "evaluation.compare.tabs.workflowDiff": "Workflow diff", + "evaluation.compare.datasetMismatch": "Versions ran against different case counts ({counts}). Comparison is aligned by row index.", + "evaluation.compare.cases.col.index": "#", + "evaluation.compare.cases.col.input": "Case input", + "evaluation.compare.cases.col.best": "Best", + "evaluation.compare.cases.col.deltaVsBest": "Δ vs best", + "evaluation.compare.cases.bestPill": "{letter} ★", + "evaluation.compare.cases.loading": "Loading cases…", + "evaluation.compare.cases.running": "Evaluation in progress — scores fill in as each case completes.", + "evaluation.compare.cases.empty": "No case-level results to compare yet.", + "evaluation.compare.cases.loadError": "Some versions' case data couldn't be loaded.", + "evaluation.compare.outputs.casesSidebarTitle": "Test cases", + "evaluation.compare.outputs.input": "Input", + "evaluation.compare.outputs.noOutput": "No output for this version.", + "evaluation.compare.metrics.col.metric": "Metric", + "evaluation.compare.metrics.empty": "No score-shaped metrics to compare yet.", + "evaluation.metric.description.correctness": "Whether the answer’s meaning matches a reference answer. Scored 1 (worst)–5 (best).", + "evaluation.metric.description.helpfulness": "Whether the response addresses the query. Scored 1 (worst)–5 (best).", + "evaluation.metric.description.stringSimilarity": "How close the answer is to a reference answer, character by character. Scored 0–1.", + "evaluation.metric.description.categorization": "Whether the answer exactly matches the reference answer. 1 if so, 0 otherwise.", + "evaluation.metric.description.toolsUsed": "Whether the expected tool(s) were used. Scored 0–1.", + "evaluation.metric.criteria.label": "Criteria:", + "evaluation.metric.criteria.showMore": "Show more", + "evaluation.metric.criteria.showLess": "Show less", + "evaluation.compare.workflowDiff.needTwo": "Add at least two versions to this collection to compare their workflows.", + "evaluation.compare.workflowDiff.loadError": "Could not load the workflows to compare.", + "evaluation.compare.workflowDiff.base": "Base", + "evaluation.compare.workflowDiff.compare": "Compare", "evaluation.setup.title": "New collection", "evaluation.setup.subtitle": "Pick a dataset and the versions you want to compare.", "evaluation.setup.collectionName": "Collection name", diff --git a/packages/frontend/editor-ui/src/__tests__/defaults.ts b/packages/frontend/editor-ui/src/__tests__/defaults.ts index ff8ee5f3e50..128f1db97c0 100644 --- a/packages/frontend/editor-ui/src/__tests__/defaults.ts +++ b/packages/frontend/editor-ui/src/__tests__/defaults.ts @@ -183,6 +183,7 @@ export const defaultSettings: FrontendSettings = { }, evaluation: { quota: 0, + collectionsEnabled: false, }, activeModules: [], canvasOnly: false, diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CasesTable.test.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CasesTable.test.ts new file mode 100644 index 00000000000..9efce8dfa9c --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CasesTable.test.ts @@ -0,0 +1,123 @@ +import { fireEvent } from '@testing-library/vue'; +import { describe, expect, it } from 'vitest'; + +import { createComponentRenderer } from '@/__tests__/render'; + +import type { CompareCaseCell, CompareCaseRow } from '../../composables/useCompareCases'; +import type { CompareVersion } from '../../composables/useCompareData'; +import CasesTable from './CasesTable.vue'; + +const versions: CompareVersion[] = [ + { + index: 0, + testRunId: 'run-a', + workflowVersionId: 'v0', + letter: 'A', + label: 'v0', + status: 'completed', + avgScore: null, + }, + { + index: 1, + testRunId: 'run-b', + workflowVersionId: 'v1', + letter: 'B', + label: 'v1', + status: 'completed', + avgScore: null, + }, +]; + +// A null score with no `testCaseId` represents a case this version's run never +// covered (dataset drift → `⊘`); a null score with a `testCaseId` is a case +// still running (→ `–`). +const cell = (versionIndex: number, score: number | null): CompareCaseCell => ({ + versionIndex, + testCaseId: score === null ? null : `c${versionIndex}`, + inputs: { q: 'x' }, + outputs: { output: 'y' }, + metrics: score === null ? undefined : { helpfulness: score }, + score, +}); + +const row = (index: number, scores: Array, best: number | null): CompareCaseRow => ({ + index, + displayIndex: index + 1, + inputPreview: `case ${index}`, + cells: scores.map((s, i) => cell(i, s)), + bestVersionIndex: best, +}); + +const renderComponent = createComponentRenderer(CasesTable); + +describe('CasesTable', () => { + it('renders one row per case with the best-version pill', () => { + const { container } = renderComponent({ + props: { versions, caseRows: [row(0, [0.7, 0.9], 1), row(1, [0.6, 0.5], 0)] }, + }); + + expect(container.querySelectorAll('[data-test-id="compare-cases-row"]')).toHaveLength(2); + // best pills render the winning version's letter + expect(container.textContent).toContain('B ★'); + expect(container.textContent).toContain('A ★'); + }); + + it('sorts by score spread descending by default (biggest regression first)', () => { + const { container } = renderComponent({ + props: { + versions, + // row 0 spread 0.05, row 1 spread 0.4 → row 1 should come first + caseRows: [row(0, [0.9, 0.85], 0), row(1, [0.9, 0.5], 0)], + }, + }); + + const rows = container.querySelectorAll('[data-test-id="compare-cases-row"]'); + // first rendered row is the bigger-spread case (#2) + expect(rows[0].textContent).toContain('2'); + }); + + it('emits drilldown with the case index when a row is clicked', async () => { + const { container, emitted } = renderComponent({ + props: { versions, caseRows: [row(3, [0.7, 0.9], 1)] }, + }); + + await fireEvent.click(container.querySelector('[data-test-id="compare-cases-row"]')!); + expect(emitted().drilldown).toEqual([[3]]); + }); + + it('renders a missing-case marker (⊘) for a case absent from a version', () => { + const { container } = renderComponent({ + props: { versions, caseRows: [row(0, [0.7, null], 0)] }, + }); + + expect(container.textContent).toContain('⊘'); + }); + + it('renders a pending marker (–) for a scored-yet case that still exists', () => { + const pendingCell: CompareCaseCell = { + versionIndex: 1, + testCaseId: 'c1', + inputs: {}, + outputs: undefined, + metrics: undefined, + score: null, + }; + const { container } = renderComponent({ + props: { + versions, + caseRows: [ + { + index: 0, + displayIndex: 1, + inputPreview: 'case 0', + cells: [cell(0, 0.7), pendingCell], + bestVersionIndex: 0, + }, + ], + }, + }); + + expect(container.textContent).toContain('–'); + expect(container.textContent).not.toContain('⊘'); + }); +}); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CasesTable.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CasesTable.vue new file mode 100644 index 00000000000..96a8fcbd227 --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CasesTable.vue @@ -0,0 +1,246 @@ + + + + + diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareHeader.test.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareHeader.test.ts index 0e5d485738b..47e1e967727 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareHeader.test.ts +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareHeader.test.ts @@ -56,4 +56,20 @@ describe('CompareHeader', () => { expect(container.textContent).toContain('Running'); }); + + it('shows the failed badge when every run errored', () => { + const { container } = renderComponent({ + props: { + collectionName: 'Exp', + versions: [ + version({ index: 0, status: 'error', avgScore: null }), + version({ index: 1, status: 'cancelled', avgScore: null }), + ], + bestVersionIndex: null, + }, + }); + + expect(container.textContent).toContain('Failed'); + expect(container.textContent).not.toContain('Done'); + }); }); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareHeader.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareHeader.vue index 2b69d4249cb..f6864a45f65 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareHeader.vue +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareHeader.vue @@ -17,6 +17,26 @@ const i18n = useI18n(); const status = computed(() => deriveRunsStatus(props.versions)); +const statusBadge = computed(() => { + switch (status.value) { + case 'error': + return { + theme: 'warning' as const, + label: i18n.baseText('evaluation.collections.card.failed'), + }; + case 'running': + return { + theme: 'tertiary' as const, + label: i18n.baseText('evaluation.collections.card.running'), + }; + default: + return { + theme: 'success' as const, + label: i18n.baseText('evaluation.collections.card.done'), + }; + } +}); + const legend = computed(() => props.versions.map((version) => ({ ...version, @@ -30,15 +50,7 @@ const legend = computed(() =>
{{ collectionName }} - - {{ - i18n.baseText( - status === 'done' - ? 'evaluation.collections.card.done' - : 'evaluation.collections.card.running', - ) - }} - + {{ statusBadge.label }}
{{ diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareTabs.test.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareTabs.test.ts new file mode 100644 index 00000000000..47f5e62694a --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareTabs.test.ts @@ -0,0 +1,92 @@ +import { fireEvent, waitFor } from '@testing-library/vue'; +import { describe, expect, it } from 'vitest'; + +import { createComponentRenderer } from '@/__tests__/render'; + +import type { CompareCaseRow } from '../../composables/useCompareCases'; +import type { CompareMetricGroup, CompareVersion } from '../../composables/useCompareData'; +import CompareTabs from './CompareTabs.vue'; + +const versions: CompareVersion[] = [ + { + index: 0, + testRunId: 'run-a', + workflowVersionId: 'v0', + letter: 'A', + label: 'v0', + status: 'completed', + avgScore: null, + }, + { + index: 1, + testRunId: 'run-b', + workflowVersionId: 'v1', + letter: 'B', + label: 'v1', + status: 'completed', + avgScore: null, + }, +]; + +const metricGroups: CompareMetricGroup[] = [ + { key: 'helpfulness', label: 'Helpfulness', values: [0.7, 0.9], bestIndex: 1 }, +]; + +const caseRows: CompareCaseRow[] = [ + { + index: 0, + displayIndex: 1, + inputPreview: 'case 0', + cells: [ + { + versionIndex: 0, + testCaseId: 'a', + inputs: {}, + outputs: { output: 'x' }, + metrics: { helpfulness: 0.7 }, + score: 0.7, + }, + { + versionIndex: 1, + testCaseId: 'b', + inputs: {}, + outputs: { output: 'y' }, + metrics: { helpfulness: 0.9 }, + score: 0.9, + }, + ], + bestVersionIndex: 1, + }, +]; + +const renderComponent = createComponentRenderer(CompareTabs); + +describe('CompareTabs', () => { + it('shows the Cases tab by default', () => { + const { container } = renderComponent({ + props: { versions, metricGroups, caseRows, casesLoading: false }, + }); + + expect(container.querySelector('[data-test-id="compare-cases-table"]')).not.toBeNull(); + }); + + it('shows a loading message while cases load', () => { + const { container } = renderComponent({ + props: { versions, metricGroups, caseRows: [], casesLoading: true }, + }); + + expect(container.querySelector('[data-test-id="compare-cases-table"]')).toBeNull(); + expect(container.textContent).toContain('Loading cases'); + }); + + it('drilling into a case row switches to the Outputs tab', async () => { + const { container } = renderComponent({ + props: { versions, metricGroups, caseRows, casesLoading: false }, + }); + + await fireEvent.click(container.querySelector('[data-test-id="compare-cases-row"]')!); + await waitFor(() => + expect(container.querySelector('[data-test-id="compare-outputs-tab"]')).not.toBeNull(), + ); + }); +}); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareTabs.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareTabs.vue new file mode 100644 index 00000000000..2286ade96a8 --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/CompareTabs.vue @@ -0,0 +1,127 @@ + + + + + diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/DatasetMismatchBanner.test.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/DatasetMismatchBanner.test.ts new file mode 100644 index 00000000000..b759a6acf73 --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/DatasetMismatchBanner.test.ts @@ -0,0 +1,19 @@ +import { describe, expect, it } from 'vitest'; + +import { createComponentRenderer } from '@/__tests__/render'; + +import DatasetMismatchBanner from './DatasetMismatchBanner.vue'; + +const renderComponent = createComponentRenderer(DatasetMismatchBanner); + +describe('DatasetMismatchBanner', () => { + it('renders the per-version case counts in the warning', () => { + const { container } = renderComponent({ + props: { mismatch: { hasMismatch: true, counts: [12, 12, 10], maxCount: 12 } }, + }); + + const banner = container.querySelector('[data-test-id="compare-dataset-mismatch"]'); + expect(banner).not.toBeNull(); + expect(banner?.textContent).toContain('12, 12, 10'); + }); +}); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/DatasetMismatchBanner.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/DatasetMismatchBanner.vue new file mode 100644 index 00000000000..5cc7ac76f47 --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/DatasetMismatchBanner.vue @@ -0,0 +1,25 @@ + + + diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/MetricCriteria.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/MetricCriteria.vue new file mode 100644 index 00000000000..b5d491827b6 --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/MetricCriteria.vue @@ -0,0 +1,97 @@ + + + + + diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/MetricsTab.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/MetricsTab.vue new file mode 100644 index 00000000000..2b102ad90fd --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/MetricsTab.vue @@ -0,0 +1,102 @@ + + + + + diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/OutputsTab.test.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/OutputsTab.test.ts new file mode 100644 index 00000000000..0f4c66e8e99 --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/OutputsTab.test.ts @@ -0,0 +1,79 @@ +import { fireEvent } from '@testing-library/vue'; +import { describe, expect, it } from 'vitest'; + +import { createComponentRenderer } from '@/__tests__/render'; + +import type { CompareCaseCell, CompareCaseRow } from '../../composables/useCompareCases'; +import type { CompareVersion } from '../../composables/useCompareData'; +import OutputsTab from './OutputsTab.vue'; + +const versions: CompareVersion[] = [ + { + index: 0, + testRunId: 'run-a', + workflowVersionId: 'v0', + letter: 'A', + label: 'Baseline', + status: 'completed', + avgScore: null, + }, + { + index: 1, + testRunId: 'run-b', + workflowVersionId: 'v1', + letter: 'B', + label: 'Candidate', + status: 'completed', + avgScore: null, + }, +]; + +const cell = (versionIndex: number, output: string): CompareCaseCell => ({ + versionIndex, + testCaseId: `c${versionIndex}`, + inputs: { q: 'What is 2+2?' }, + outputs: { output }, + metrics: { helpfulness: 0.8 }, + score: 0.8, +}); + +const rows: CompareCaseRow[] = [ + { + index: 0, + displayIndex: 1, + inputPreview: 'What is 2+2?', + cells: [cell(0, 'four'), cell(1, 'the answer is four')], + bestVersionIndex: 1, + }, + { + index: 1, + displayIndex: 2, + inputPreview: 'Capital of France?', + cells: [cell(0, 'Paris'), cell(1, 'Paris, France')], + bestVersionIndex: 1, + }, +]; + +const renderComponent = createComponentRenderer(OutputsTab); + +describe('OutputsTab', () => { + it('renders one output column per version for the selected case', () => { + const { container } = renderComponent({ + props: { versions, caseRows: rows, selectedIndex: 0 }, + }); + + expect(container.querySelectorAll('[data-test-id="compare-outputs-column"]')).toHaveLength(2); + expect(container.textContent).toContain('four'); + expect(container.textContent).toContain('the answer is four'); + }); + + it('emits the new index when a sidebar case is clicked', async () => { + const { container, emitted } = renderComponent({ + props: { versions, caseRows: rows, selectedIndex: 0 }, + }); + + const items = container.querySelectorAll('[data-test-id="compare-outputs-case"]'); + await fireEvent.click(items[1]); + expect(emitted()['update:selectedIndex']).toEqual([[1]]); + }); +}); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/OutputsTab.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/OutputsTab.vue new file mode 100644 index 00000000000..dc7334d3eb7 --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/OutputsTab.vue @@ -0,0 +1,220 @@ + + + + + diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/ScoreChart.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/ScoreChart.vue index 166ba6cc3b6..40ba8459230 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/ScoreChart.vue +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/ScoreChart.vue @@ -5,10 +5,13 @@ import { computed, ref } from 'vue'; import type { CompareMetricGroup, CompareVersion } from '../../composables/useCompareData'; import GroupedMetricChart from '../shared/GroupedMetricChart.vue'; +import MetricCriteria from './MetricCriteria.vue'; const props = defineProps<{ metricGroups: CompareMetricGroup[]; versions: CompareVersion[]; + // metric name → its custom LLM-judge prompt, when configured. + metricPrompts?: Record; }>(); const i18n = useI18n(); @@ -59,6 +62,7 @@ const letters = computed(() => props.versions.map((version) => version.letter)); {{ group.label }} + ({ + useWorkflowsListStore: () => ({ fetchWorkflow }), +})); +vi.mock('@/features/workflows/workflowHistory/workflowHistory.store', () => ({ + useWorkflowHistoryStore: () => ({ getWorkflowVersion }), +})); +vi.mock('@/app/composables/useToast', () => ({ + useToast: () => ({ showError: vi.fn() }), +})); +// Stub the heavy diff canvas — this tab only wires data into it. +vi.mock('@/features/workflows/workflowDiff/WorkflowDiffView.vue', () => ({ + default: { name: 'WorkflowDiffView', template: '
' }, +})); + +const version = (over: Partial): CompareVersion => ({ + index: 0, + testRunId: 'run', + workflowVersionId: 'v0', + letter: 'A', + label: 'Baseline', + status: 'completed', + avgScore: null, + ...over, +}); + +const renderComponent = createComponentRenderer(WorkflowDiffTab); + +describe('WorkflowDiffTab', () => { + beforeEach(() => { + fetchWorkflow.mockReset().mockResolvedValue({ id: 'wf-1', nodes: [], connections: {} }); + getWorkflowVersion.mockReset().mockResolvedValue({ + versionId: 'v1', + workflowId: 'wf-1', + nodes: [], + connections: {}, + nodeGroups: [], + }); + }); + + it('prompts for a second version and skips loading when only one is present', () => { + const { queryByTestId } = renderComponent({ + props: { versions: [version({ index: 0 })], workflowId: 'wf-1' }, + }); + + expect(queryByTestId('workflow-diff-source-select')).toBeNull(); + expect(fetchWorkflow).not.toHaveBeenCalled(); + }); + + it('resolves the current draft from the base workflow and a version from its snapshot', async () => { + const versions = [ + version({ index: 0, workflowVersionId: null, letter: 'A', label: 'Current draft' }), + version({ index: 1, workflowVersionId: 'v1', letter: 'B', label: 'v1' }), + ]; + + const { findByTestId } = renderComponent({ props: { versions, workflowId: 'wf-1' } }); + + await findByTestId('wf-diff-stub'); + await waitFor(() => expect(fetchWorkflow).toHaveBeenCalledWith('wf-1')); + // Only the real version pulls a history snapshot; the draft reuses the base. + expect(getWorkflowVersion).toHaveBeenCalledTimes(1); + expect(getWorkflowVersion).toHaveBeenCalledWith('wf-1', 'v1'); + }); +}); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/WorkflowDiffTab.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/WorkflowDiffTab.vue new file mode 100644 index 00000000000..563d5e3544b --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/Compare/WorkflowDiffTab.vue @@ -0,0 +1,227 @@ + + + + + diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/EvalCollectionsListView/CollectionCard.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/EvalCollectionsListView/CollectionCard.vue index 3a4232601a4..a5e4bdaf72d 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/EvalCollectionsListView/CollectionCard.vue +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/EvalCollectionsListView/CollectionCard.vue @@ -36,10 +36,33 @@ const openCompare = () => { // `null` until the detail (with run statuses) has loaded — the list view only // pre-fetches detail for the first few cards and lazy-loads the rest on hover, // so we must not assert "Done" for a card whose runs might still be in flight. -const status = computed<'done' | 'running' | null>(() => +const status = computed<'done' | 'running' | 'error' | null>(() => props.detail ? deriveRunsStatus(props.detail.runs) : null, ); +// Badge theme + label per status; `null` while detail is still loading (no badge). +const statusBadge = computed(() => { + switch (status.value) { + case 'done': + return { + theme: 'success' as const, + label: i18n.baseText('evaluation.collections.card.done'), + }; + case 'running': + return { + theme: 'tertiary' as const, + label: i18n.baseText('evaluation.collections.card.running'), + }; + case 'error': + return { + theme: 'warning' as const, + label: i18n.baseText('evaluation.collections.card.failed'), + }; + default: + return null; + } +}); + // Append a right arrow so the CTA reads "Open compare →" the way the // Figma mock does. N8nButton doesn't accept a trailing icon prop today, // so the arrow lives in the label string. @@ -145,14 +168,8 @@ onMounted(() => observe(cardRef.value));
{{ collection.name }} - - {{ - i18n.baseText( - status === 'done' - ? 'evaluation.collections.card.done' - : 'evaluation.collections.card.running', - ) - }} + + {{ statusBadge.label }}
diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/EvalCollectionsListView/UngroupedRunRow.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/EvalCollectionsListView/UngroupedRunRow.vue index aa93029fe10..91f3430f20e 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/EvalCollectionsListView/UngroupedRunRow.vue +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/EvalCollectionsListView/UngroupedRunRow.vue @@ -4,7 +4,7 @@ import { useI18n, type BaseTextKey } from '@n8n/i18n'; import { computed } from 'vue'; import type { TestRunRecord } from '../../evaluation.api'; -import { isScoreShapedMetric } from '../../evaluation.utils'; +import { averageNormalizedScore } from '../../evaluation.utils'; const STATUS_PILL_THEME: Record = { completed: 'success', @@ -35,16 +35,13 @@ const props = defineProps<{ const i18n = useI18n(); -// Average only score-shaped metrics (values in [0, 1]). Eval-config metrics -// commonly co-exist with absolute counts (tokens, latency_ms) in the same -// `metrics` map, and naively averaging across all of them produces nonsense -// like `198431%` (mostly the token total). +// Average the run's score metrics, each normalized to [0, 1] by its scale. +// Eval-config metrics commonly co-exist with absolute counts (tokens, +// latency_ms) in the same `metrics` map; those aren't scores and are dropped, +// so a naive all-metric average can't produce nonsense like `198431%`. const score = computed(() => { - const m = props.run.metrics; - if (!m) return null; - const values = Object.values(m).filter(isScoreShapedMetric); - if (values.length === 0) return null; - return Math.round((values.reduce((a, b) => a + b, 0) / values.length) * 100); + const avg = averageNormalizedScore(props.run.metrics); + return avg === null ? null : Math.round(avg * 100); }); const statusTheme = computed(() => STATUS_PILL_THEME[props.run.status] ?? 'tertiary'); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/ListRuns/RunsSection.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/ListRuns/RunsSection.vue index 2914e4be73d..f10258cb9c7 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/ListRuns/RunsSection.vue +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/ListRuns/RunsSection.vue @@ -2,15 +2,19 @@ import type { TestRunRecord } from '../../evaluation.api'; import MetricsChart from './MetricsChart.vue'; import TestRunsTable from './TestRunsTable.vue'; +import { N8nPagination } from '@n8n/design-system'; import { useI18n } from '@n8n/i18n'; import { VIEWS } from '@/app/constants'; import { convertToDisplayDate } from '@/app/utils/formatters/dateFormatter'; -import { computed } from 'vue'; +import { computed, ref, watch } from 'vue'; import { useRouter } from 'vue-router'; const props = defineProps<{ runs: Array; workflowId: string; + // When set, the table paginates to this many rows per page. The chart still + // plots every run. Omitted → the full table renders, no pagination. + pageSize?: number; }>(); const locale = useI18n(); @@ -18,6 +22,35 @@ const router = useRouter(); const selectedMetric = defineModel('selectedMetric', { required: true }); +const currentPage = ref(1); +const showPagination = computed(() => !!props.pageSize && props.runs.length > props.pageSize); + +// Newest-first ordering for pagination so page 1 holds the most recent runs +// (the incoming `runs` prop is ascending, for the chart's left→right trend). +// Ties on `runAt` fall back to run number so equal-timestamp runs stay in +// sequence. The table re-applies its own descending sort within each page. +const runsNewestFirst = computed(() => + [...props.runs].sort((a, b) => { + const byDate = new Date(b.runAt).getTime() - new Date(a.runAt).getTime(); + return byDate !== 0 ? byDate : b.index - a.index; + }), +); +const pagedRuns = computed(() => { + if (!props.pageSize) return props.runs; + const start = (currentPage.value - 1) * props.pageSize; + return runsNewestFirst.value.slice(start, start + props.pageSize); +}); + +// Clamp the page when the run set shrinks (or polling changes it) so we never +// land on an empty page past the end. +watch( + () => props.runs.length, + () => { + const maxPage = props.pageSize ? Math.max(1, Math.ceil(props.runs.length / props.pageSize)) : 1; + if (currentPage.value > maxPage) currentPage.value = maxPage; + }, +); + const metrics = computed(() => { const metricKeys = props.runs.reduce((acc, run) => { Object.keys(run.metrics ?? {}).forEach((metric) => acc.add(metric)); @@ -74,17 +107,27 @@ const handleRowClick = (row: TestRunRecord) => { @@ -97,4 +140,18 @@ const handleRowClick = (row: TestRunRecord) => { overflow: auto; margin-bottom: 20px; } + +// Stacked/paginated mode: the table is capped to `pageSize` rows so it never +// needs its own scroll region. Let the section grow with its content and defer +// scrolling to the page so there's a single page-level scrollbar. +.paged { + flex: none; + overflow: visible; + margin-bottom: 0; +} + +.pagination { + display: flex; + justify-content: center; +} diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/shared/GroupedMetricChart.test.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/shared/GroupedMetricChart.test.ts index f58b0181b28..36f949a2108 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/shared/GroupedMetricChart.test.ts +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/shared/GroupedMetricChart.test.ts @@ -48,6 +48,6 @@ describe('GroupedMetricChart', () => { // 0.9 ≥ 0.6 keeps its version color; 0.4 < 0.6 flips to danger. expect(bars[0].getAttribute('fill')).toBe(versionColorVar(0)); - expect(bars[1].getAttribute('fill')).toBe('var(--color--red-700)'); + expect(bars[1].getAttribute('fill')).toBe('var(--icon-color--danger)'); }); }); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/shared/GroupedMetricChart.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/shared/GroupedMetricChart.vue index e64d100de67..c225b81c7e8 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/shared/GroupedMetricChart.vue +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/components/shared/GroupedMetricChart.vue @@ -47,7 +47,9 @@ const GEOMETRY = { const geo = computed(() => GEOMETRY[props.variant]); -const CRITICAL_COLOR = 'var(--color--red-700)'; +// Semantic danger token (not the `--color--red-*` primitive scale): it adapts +// to dark theme and stays distinct from the version palette's red hue. +const CRITICAL_COLOR = 'var(--icon-color--danger)'; const clamp01 = (v: number, max: number) => { if (!Number.isFinite(v)) return 0; diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareCases.test.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareCases.test.ts new file mode 100644 index 00000000000..42206b7956f --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareCases.test.ts @@ -0,0 +1,193 @@ +import { createTestingPinia } from '@pinia/testing'; +import { setActivePinia } from 'pinia'; +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { nextTick, ref } from 'vue'; + +import type { TestCaseExecutionRecord } from '../evaluation.api'; +import { useEvaluationStore } from '../evaluation.store'; +import type { EvaluationCollectionDetail } from '../evalCollections.types'; +import { useCompareCases } from './useCompareCases'; + +const run = (testRunId: string): EvaluationCollectionDetail['runs'][number] => ({ + testRunId, + workflowVersionId: testRunId, + status: 'completed', + runAt: null, + completedAt: null, + avgScore: null, + metrics: null, +}); + +const detailWith = (runIds: string[]): EvaluationCollectionDetail => ({ + id: 'col-1', + name: 'Compare', + description: null, + workflowId: 'wf-1', + evaluationConfigId: 'cfg-1', + createdById: 'u1', + createdAt: '', + updatedAt: '', + runCount: runIds.length, + runs: runIds.map(run), +}); + +const caseRecord = ( + overrides: Partial & Pick, +): TestCaseExecutionRecord => ({ + executionId: null, + status: 'success', + createdAt: '', + updatedAt: '', + runAt: null, + ...overrides, +}); + +describe('useCompareCases', () => { + let store: ReturnType; + + beforeEach(() => { + setActivePinia(createTestingPinia({ stubActions: false, createSpy: vi.fn })); + store = useEvaluationStore(); + store.fetchTestCaseExecutions = vi.fn( + async () => [], + ) as unknown as typeof store.fetchTestCaseExecutions; + }); + + function seed(records: TestCaseExecutionRecord[]) { + store.$patch((state) => { + state.testCaseExecutionsById = Object.fromEntries(records.map((r) => [r.id, r])); + }); + } + + it('aligns cases across runs by runIndex and picks the best version per case', async () => { + // helpfulness is a 1–5 AI-judge metric → normalized /5 (3.5 → 0.7 etc.). + seed([ + caseRecord({ id: 'a0', testRunId: 'run-a', runIndex: 0, metrics: { helpfulness: 3.5 } }), + caseRecord({ id: 'a1', testRunId: 'run-a', runIndex: 1, metrics: { helpfulness: 2.5 } }), + // run-b returned out of order — alignment must sort by runIndex. + caseRecord({ id: 'b1', testRunId: 'run-b', runIndex: 1, metrics: { helpfulness: 4.5 } }), + caseRecord({ id: 'b0', testRunId: 'run-b', runIndex: 0, metrics: { helpfulness: 3 } }), + ]); + const { caseRows, mismatch } = useCompareCases( + ref(detailWith(['run-a', 'run-b'])), + ref('wf-1'), + ); + await nextTick(); + + expect(mismatch.value.hasMismatch).toBe(false); + expect(caseRows.value).toHaveLength(2); + // case #0: A 0.7 vs B 0.6 → A best + expect(caseRows.value[0].bestVersionIndex).toBe(0); + expect(caseRows.value[0].cells[1].testCaseId).toBe('b0'); + // case #1: A 0.5 vs B 0.9 → B best + expect(caseRows.value[1].bestVersionIndex).toBe(1); + expect(caseRows.value[1].cells.map((c) => c.score)).toEqual([0.5, 0.9]); + }); + + it('flags a dataset mismatch and null-fills missing cells when case counts diverge', async () => { + // helpfulness is a 1–5 AI-judge metric → normalized /5 (4 → 0.8 etc.). + seed([ + caseRecord({ id: 'a0', testRunId: 'run-a', runIndex: 0, metrics: { helpfulness: 3.5 } }), + caseRecord({ id: 'a1', testRunId: 'run-a', runIndex: 1, metrics: { helpfulness: 4 } }), + caseRecord({ id: 'b0', testRunId: 'run-b', runIndex: 0, metrics: { helpfulness: 3 } }), + ]); + const { caseRows, mismatch } = useCompareCases( + ref(detailWith(['run-a', 'run-b'])), + ref('wf-1'), + ); + await nextTick(); + + expect(mismatch.value.hasMismatch).toBe(true); + expect(mismatch.value.counts).toEqual([2, 1]); + // run-b has no case #1 → its cell is null-filled, run-a keeps its value. + expect(caseRows.value[1].cells[0].score).toBe(0.8); + expect(caseRows.value[1].cells[1].testCaseId).toBeNull(); + expect(caseRows.value[1].cells[1].score).toBeNull(); + }); + + it('fans out fetchTestCaseExecutions once per run and toggles loading', async () => { + const fetchSpy = vi.fn(async ({ runId }: { workflowId: string; runId: string }) => { + store.$patch((state) => { + state.testCaseExecutionsById = { + ...state.testCaseExecutionsById, + [`${runId}-0`]: caseRecord({ id: `${runId}-0`, testRunId: runId, runIndex: 0 }), + }; + }); + return []; + }); + store.fetchTestCaseExecutions = fetchSpy as unknown as typeof store.fetchTestCaseExecutions; + + const { loading, casesLoaded, caseRows, load } = useCompareCases( + ref(detailWith(['run-a', 'run-b'])), + ref('wf-1'), + ); + await load(); + + expect(fetchSpy).toHaveBeenCalledWith({ workflowId: 'wf-1', runId: 'run-a' }); + expect(fetchSpy).toHaveBeenCalledWith({ workflowId: 'wf-1', runId: 'run-b' }); + expect(loading.value).toBe(false); + expect(casesLoaded.value).toBe(true); + expect(caseRows.value).toHaveLength(1); + }); + + it('flags casesError when a run fetch rejects (not a real mismatch)', async () => { + store.fetchTestCaseExecutions = vi.fn(async ({ runId }: { runId: string }) => { + if (runId === 'run-b') throw new Error('network'); + return []; + }) as unknown as typeof store.fetchTestCaseExecutions; + + const { casesError, load } = useCompareCases(ref(detailWith(['run-a', 'run-b'])), ref('wf-1')); + await load(); + + expect(casesError.value).toBe(true); + }); + + it('excludes predefined operational metrics from the per-case score', async () => { + seed([ + caseRecord({ + id: 'a0', + testRunId: 'run-a', + runIndex: 0, + metrics: { helpfulness: 4, totalTokens: 1, executionTime: 0.4 }, + }), + ]); + const { caseRows } = useCompareCases(ref(detailWith(['run-a'])), ref('wf-1')); + await nextTick(); + + // score is the mean of normalized score metrics → just helpfulness (4/5 = 0.8); + // tokens/execution time are operational and excluded. + expect(caseRows.value[0].cells[0].score).toBe(0.8); + }); + + it('aligns by runIndex so a version missing a middle case does not shift later cases', async () => { + seed([ + caseRecord({ id: 'a0', testRunId: 'run-a', runIndex: 0, metrics: { helpfulness: 5 } }), + caseRecord({ id: 'a1', testRunId: 'run-a', runIndex: 1, metrics: { helpfulness: 4 } }), + caseRecord({ id: 'a2', testRunId: 'run-a', runIndex: 2, metrics: { helpfulness: 3 } }), + // run-b is missing the middle case (runIndex 1). + caseRecord({ id: 'b0', testRunId: 'run-b', runIndex: 0, metrics: { helpfulness: 5 } }), + caseRecord({ id: 'b2', testRunId: 'run-b', runIndex: 2, metrics: { helpfulness: 2 } }), + ]); + const { caseRows } = useCompareCases(ref(detailWith(['run-a', 'run-b'])), ref('wf-1')); + await nextTick(); + + expect(caseRows.value).toHaveLength(3); + // runIndex 1: run-b has no case → null cell, not run-b's next case shifted up. + expect(caseRows.value[1].cells[0].testCaseId).toBe('a1'); + expect(caseRows.value[1].cells[1].testCaseId).toBeNull(); + // runIndex 2: b2 stays paired with a2 (same seeded case), not with a1. + expect(caseRows.value[2].cells[0].testCaseId).toBe('a2'); + expect(caseRows.value[2].cells[1].testCaseId).toBe('b2'); + }); + + it('resolves to a loaded (not stuck) state for an empty collection', async () => { + const { casesLoaded, loading, caseRows } = useCompareCases(ref(detailWith([])), ref('wf-1')); + await nextTick(); + + // The watcher must still invoke load() for a detail with no runs so its + // empty-run completion branch runs, rather than sitting in loading forever. + expect(casesLoaded.value).toBe(true); + expect(loading.value).toBe(false); + expect(caseRows.value).toEqual([]); + }); +}); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareCases.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareCases.ts new file mode 100644 index 00000000000..71543fb3aa1 --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareCases.ts @@ -0,0 +1,195 @@ +import orderBy from 'lodash/orderBy'; +import type { JsonObject } from 'n8n-workflow'; +import { computed, ref, watch, type Ref } from 'vue'; + +import type { TestCaseExecutionRecord } from '../evaluation.api'; +import { useEvaluationStore } from '../evaluation.store'; +import type { EvaluationCollectionDetail } from '../evalCollections.types'; +import { averageNormalizedScore, indexOfMax, stringifyValue } from '../evaluation.utils'; + +// One version's execution of a single aligned test case. `null`-valued fields +// mark a case that this version's run didn't cover (dataset drift). +export interface CompareCaseCell { + versionIndex: number; + testCaseId: string | null; + inputs: JsonObject | undefined; + outputs: JsonObject | undefined; + metrics: Record | undefined; + // Mean of the case's score-shaped ([0, 1], non-predefined) metrics, or null + // when the version skipped this case or reported no score-shaped metric. + score: number | null; +} + +export interface CompareCaseRow { + index: number; + displayIndex: number; + inputPreview: string; + cells: CompareCaseCell[]; + // Version index with the highest case score, or null if none scored. + bestVersionIndex: number | null; +} + +export interface DatasetMismatch { + hasMismatch: boolean; + // Case count per version, aligned to run order. + counts: number[]; + maxCount: number; +} + +// Compact one-line preview of a case's inputs for the table's first column. +function inputPreview(inputs: JsonObject | undefined): string { + if (!inputs) return ''; + return Object.values(inputs) + .map((value) => stringifyValue(value)) + .filter((text) => text.length > 0) + .join(' · '); +} + +/** + * Loads per-case executions for every run in a collection and aligns them into + * one row per test case across versions. + * + * There is no collection-level per-case endpoint and no case id shared across + * runs, so this fans out `fetchTestCaseExecutions` per run and aligns cells by + * `runIndex` (the seeded per-case sequence) — a version missing a case leaves a + * null cell rather than shifting later cases into the wrong row. Divergent case + * counts surface as a `mismatch` rather than silently misaligning rows. + */ +export function useCompareCases( + detail: Ref, + workflowId: Ref, +) { + const evaluationStore = useEvaluationStore(); + + const loading = ref(false); + // True once the current run set's per-case fetches have completed at least + // once. Downstream gates (mismatch banner, telemetry) use this rather than + // `!loading` so they can't act on the empty window before the first load or + // on a superseded load's transient `loading = false`. + const casesLoaded = ref(false); + // True when any run's per-case fetch failed. Distinguishes a transient + // failure from a real dataset mismatch — a failed run also comes back with + // zero cases, which would otherwise read as "diverging case counts". + const casesError = ref(false); + + // Monotonic token so a slow load for a previous collection can't flip state + // out from under the collection the user has since switched to. + let loadToken = 0; + + async function load() { + const runs = detail.value?.runs ?? []; + const token = ++loadToken; + if (runs.length === 0) { + loading.value = false; + casesError.value = false; + casesLoaded.value = true; + return; + } + loading.value = true; + casesLoaded.value = false; + casesError.value = false; + try { + const results = await Promise.allSettled( + runs.map( + async (run) => + await evaluationStore.fetchTestCaseExecutions({ + workflowId: workflowId.value, + runId: run.testRunId, + }), + ), + ); + // A newer load for a different run set has taken over — don't clobber it. + if (token !== loadToken) return; + casesError.value = results.some((result) => result.status === 'rejected'); + casesLoaded.value = true; + } finally { + if (token === loadToken) loading.value = false; + } + } + + // Per-run, sorted case lists. Bucket the shared (app-global, poll-mutated) + // store map by `testRunId` in a single pass instead of re-scanning it per + // run, then sort each run's bucket by the same [runIndex, runAt] ordering + // the run-detail view uses so aligned positions map to the same seeded case. + const casesByVersion = computed(() => { + const runs = detail.value?.runs ?? []; + const byRunId = new Map( + runs.map((run) => [run.testRunId, []]), + ); + for (const record of Object.values(evaluationStore.testCaseExecutionsById)) { + const bucket = record.testRunId ? byRunId.get(record.testRunId) : undefined; + if (bucket) bucket.push(record); + } + return runs.map((run) => + orderBy( + byRunId.get(run.testRunId) ?? [], + [(record) => record.runIndex ?? Number.MAX_SAFE_INTEGER, (record) => record.runAt ?? ''], + ['asc', 'asc'], + ), + ); + }); + + const mismatch = computed(() => { + const counts = casesByVersion.value.map((cases) => cases.length); + const maxCount = counts.length ? Math.max(...counts) : 0; + return { + counts, + maxCount, + hasMismatch: counts.some((count) => count !== maxCount), + }; + }); + + const caseRows = computed(() => { + // Align cells by `runIndex` (the seeded per-case sequence), not list + // position: a version missing a case in the middle must leave a null cell + // in that row rather than shift every later case up and pair unrelated + // inputs. Fall back to list position only when a record has no runIndex. + const byIndex = casesByVersion.value.map((cases) => { + const map = new Map(); + cases.forEach((record, position) => map.set(record.runIndex ?? position, record)); + return map; + }); + + const allIndices = [...new Set(byIndex.flatMap((map) => [...map.keys()]))].sort( + (a, b) => a - b, + ); + + return allIndices.map((runIndex, rowIndex) => { + const cells: CompareCaseCell[] = byIndex.map((map, versionIndex) => { + const record = map.get(runIndex); + return { + versionIndex, + testCaseId: record?.id ?? null, + inputs: record?.inputs, + outputs: record?.outputs, + metrics: record?.metrics, + score: averageNormalizedScore(record?.metrics), + }; + }); + + const firstWithInputs = cells.find((cell) => cell.inputs !== undefined); + return { + index: rowIndex, + displayIndex: rowIndex + 1, + inputPreview: inputPreview(firstWithInputs?.inputs), + cells, + bestVersionIndex: indexOfMax(cells.map((cell) => cell.score)), + }; + }); + }); + + // Refetch whenever the run set changes (collection switch reuses the view). + // The key is `null` while the detail hasn't loaded and an empty string once + // it has but with no runs — distinguishing them so an empty collection still + // runs `load()` (which resolves its own empty-run completion state) instead + // of sitting in the initial loading state forever. + watch( + () => (detail.value ? (detail.value.runs ?? []).map((run) => run.testRunId).join(',') : null), + async (key) => { + if (key !== null) await load(); + }, + { immediate: true }, + ); + + return { caseRows, mismatch, loading, casesLoaded, casesError, load }; +} diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareData.test.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareData.test.ts index 30dd817b454..27fde4d0ee1 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareData.test.ts +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareData.test.ts @@ -62,7 +62,7 @@ describe('useCompareData', () => { expect(versions[1].label).toBe('abcdef1'); }); - it('charts only score-shaped metrics and picks the best version per metric', () => { + it('charts score metrics normalized to [0,1] and picks the best version per metric', () => { const source = ref( detail([ { @@ -72,7 +72,8 @@ describe('useCompareData', () => { runAt: null, completedAt: null, avgScore: 0.7, - metrics: { helpfulness: 0.7, totalTokens: 1200 }, + // helpfulness is a 1–5 AI-judge metric → normalized /5. + metrics: { helpfulness: 3.5, totalTokens: 1200 }, }, { testRunId: 'run-b', @@ -81,14 +82,14 @@ describe('useCompareData', () => { runAt: null, completedAt: null, avgScore: 0.9, - metrics: { helpfulness: 0.9, totalTokens: 1500 }, + metrics: { helpfulness: 4.5, totalTokens: 1500 }, }, ]), ); const { compareData } = useCompareData(source); const groups = compareData.value!.metricGroups; - // totalTokens (out of [0,1]) is excluded — only helpfulness is charted. + // totalTokens is an operational metric → excluded; helpfulness charts at /5. expect(groups).toHaveLength(1); expect(groups[0].key).toBe('helpfulness'); expect(groups[0].values).toEqual([0.7, 0.9]); @@ -108,7 +109,7 @@ describe('useCompareData', () => { avgScore: 0.7, // executionTime happens to be in [0, 1] here, but it's an absolute // operational metric — it must not be charted as a score. - metrics: { helpfulness: 0.7, executionTime: 0.4, totalTokens: 1 }, + metrics: { helpfulness: 3.5, executionTime: 0.4, totalTokens: 1 }, }, ]), ); @@ -128,7 +129,8 @@ describe('useCompareData', () => { runAt: null, completedAt: null, avgScore: 0.6, - metrics: { correctness: 0.6 }, + // correctness/helpfulness are 1–5 AI-judge metrics → normalized /5. + metrics: { correctness: 3 }, }, { testRunId: 'run-b', @@ -137,7 +139,7 @@ describe('useCompareData', () => { runAt: null, completedAt: null, avgScore: 0.8, - metrics: { correctness: 0.8, helpfulness: 0.5 }, + metrics: { correctness: 4, helpfulness: 2.5 }, }, ]), ); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareData.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareData.ts index 65c4948be39..2c2ea54797b 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareData.ts +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useCompareData.ts @@ -3,7 +3,7 @@ import { computed, type Ref } from 'vue'; import { useI18n } from '@n8n/i18n'; import type { EvalCollectionRunStatus, EvaluationCollectionDetail } from '../evalCollections.types'; -import { buildScoreShapedMetricGroups, formatMetricLabel } from '../evaluation.utils'; +import { buildScoreShapedMetricGroups, formatMetricLabel, indexOfMax } from '../evaluation.utils'; import { versionLetter } from '../components/shared/versionPalette'; // One column in the compare view: a single run positioned by `index`, which @@ -35,20 +35,6 @@ export interface CompareData { bestVersionIndex: number | null; } -// Index of the max value in `values`, ignoring nulls. Ties resolve to the -// first (left-most) version, matching the letter order users read. -function indexOfMax(values: Array): number | null { - let best: number | null = null; - let bestValue = -Infinity; - values.forEach((value, index) => { - if (value !== null && value > bestValue) { - bestValue = value; - best = index; - } - }); - return best; -} - /** * Shapes a collection's aggregate detail into the compare view's model: * one `CompareVersion` per run (in stored order) and one `CompareMetricGroup` diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useEvalCollectionsFlag.test.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useEvalCollectionsFlag.test.ts new file mode 100644 index 00000000000..7db0ffa85a8 --- /dev/null +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useEvalCollectionsFlag.test.ts @@ -0,0 +1,41 @@ +import { describe, expect, it, vi } from 'vitest'; + +import { useEvalCollectionsFlag } from './useEvalCollectionsFlag'; + +const settingsState = { collectionsEnabled: false }; +const posthogState = { enabled: false }; + +vi.mock('@/app/stores/settings.store', () => ({ + useSettingsStore: () => ({ + settings: { evaluation: { collectionsEnabled: settingsState.collectionsEnabled } }, + }), +})); + +vi.mock('@/app/stores/posthog.store', () => ({ + usePostHog: () => ({ isFeatureEnabled: () => posthogState.enabled }), +})); + +describe('useEvalCollectionsFlag', () => { + it('is enabled via the backend operator override even when PostHog is off', () => { + // The telemetry-off case: the in-browser PostHog client never initializes, + // so the flag must come from the settings-provided override. + settingsState.collectionsEnabled = true; + posthogState.enabled = false; + + expect(useEvalCollectionsFlag().value).toBe(true); + }); + + it('is enabled via the PostHog cohort flag when the override is off', () => { + settingsState.collectionsEnabled = false; + posthogState.enabled = true; + + expect(useEvalCollectionsFlag().value).toBe(true); + }); + + it('is disabled when neither signal is set', () => { + settingsState.collectionsEnabled = false; + posthogState.enabled = false; + + expect(useEvalCollectionsFlag().value).toBe(false); + }); +}); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useEvalCollectionsFlag.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useEvalCollectionsFlag.ts index 81072dd3848..61c18be64b8 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useEvalCollectionsFlag.ts +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/composables/useEvalCollectionsFlag.ts @@ -2,20 +2,28 @@ import { EVAL_COLLECTIONS_FLAG } from '@n8n/api-types'; import { computed } from 'vue'; import { usePostHog } from '@/app/stores/posthog.store'; +import { useSettingsStore } from '@/app/stores/settings.store'; /** - * Frontend gate for the eval-collections feature surface. Mirrors the - * `084_eval_collections` PostHog rollout flag that the backend consults to - * 404 the controller routes. The env override - * `N8N_EVAL_COLLECTIONS_ENABLED=true` flips PostHog to "enabled for every - * user on the running main" — useful for local + QA — without round-tripping - * the cohort layer. + * Frontend gate for the eval-collections feature surface, matching the + * `084_eval_collections` flag the backend consults to 404 the controller + * routes. It combines two independent signals: * - * Coerces PostHog's `boolean | undefined` return to a strict boolean so - * `v-if="isEvalCollectionsEnabled"` is never undefined-flickering during - * the initial flag-fetch frame. + * - `settings.evaluation.collectionsEnabled` — the backend-provided operator + * override (`N8N_EVAL_COLLECTIONS_ENABLED`). Delivered in the settings + * payload, so it works even when the in-browser PostHog client never + * initializes (telemetry off), where the flag would otherwise stay false. + * - the PostHog client flag — carries per-cohort rollout when telemetry is on. + * + * Coerces to a strict boolean so `v-if="isEvalCollectionsEnabled"` never + * undefined-flickers during the initial flag-fetch frame. */ export const useEvalCollectionsFlag = () => { const postHog = usePostHog(); - return computed(() => postHog.isFeatureEnabled(EVAL_COLLECTIONS_FLAG) === true); + const settingsStore = useSettingsStore(); + return computed( + () => + settingsStore.settings.evaluation?.collectionsEnabled === true || + postHog.isFeatureEnabled(EVAL_COLLECTIONS_FLAG) === true, + ); }; diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/evaluation.utils.test.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/evaluation.utils.test.ts index 9042bed4701..db28c16ec67 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/evaluation.utils.test.ts +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/evaluation.utils.test.ts @@ -16,6 +16,7 @@ import { getDefaultOrderedColumns, getDeltaTone, getMetricCategory, + getMetricDescriptionKey, getTestCasesColumns, getTestTableHeaders, getUserDefinedMetricNames, @@ -1451,6 +1452,28 @@ describe('utils', () => { }); }); + describe('getMetricDescriptionKey', () => { + it('returns an i18n key for each built-in metric', () => { + expect(getMetricDescriptionKey('correctness')).toBe( + 'evaluation.metric.description.correctness', + ); + expect(getMetricDescriptionKey('helpfulness')).toBe( + 'evaluation.metric.description.helpfulness', + ); + expect(getMetricDescriptionKey('stringSimilarity')).toBe( + 'evaluation.metric.description.stringSimilarity', + ); + expect(getMetricDescriptionKey('categorization')).toBe( + 'evaluation.metric.description.categorization', + ); + expect(getMetricDescriptionKey('toolsUsed')).toBe('evaluation.metric.description.toolsUsed'); + }); + it('returns null for custom/unknown metrics', () => { + expect(getMetricDescriptionKey('myCustomMetric')).toBeNull(); + expect(getMetricDescriptionKey(undefined)).toBeNull(); + }); + }); + describe('extractAnswerText', () => { it('returns null as empty string', () => { expect(extractAnswerText(null)).toBe(''); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/evaluation.utils.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/evaluation.utils.ts index 94e707b0a6b..639aac9b9a2 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/evaluation.utils.ts +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/evaluation.utils.ts @@ -1,4 +1,10 @@ import startCase from 'lodash/startCase'; +import { + normalizeMetricScore, + ONE_TO_FIVE_METRIC_KEYS, + RESERVED_METRIC_KEYS, +} from '@n8n/api-types'; +import type { BaseTextKey } from '@n8n/i18n'; import type { JsonValue } from 'n8n-workflow'; import type { IconName } from '@n8n/design-system/components/N8nIcon/icons'; import type { IExecutionResponse } from '@/features/execution/executions/executions.types'; @@ -81,12 +87,7 @@ export type MetricSource = { export const SHORT_TABLE_CELL_MIN_WIDTH = 125; const LONG_TABLE_CELL_MIN_WIDTH = 250; -const PREDEFINED_METRIC_KEYS: ReadonlySet = new Set([ - 'promptTokens', - 'completionTokens', - 'totalTokens', - 'executionTime', -]); +const PREDEFINED_METRIC_KEYS: ReadonlySet = new Set(RESERVED_METRIC_KEYS); // Excludes predefined keys (token counts, execution time) emitted by every run. export function getUserDefinedMetricNames( @@ -112,33 +113,48 @@ export function normalizeMetricValue(value: number | undefined): number | undefi return value; } -// A metric value is "score-shaped" when it lands in [0, 1] — the range the -// collection cards chart and average as a percentage. Absolute counts that -// commonly share the metrics map (tokens, latency_ms) fall outside and are -// excluded so a mini bar chart (clamped to max=1) doesn't render a bogus -// maxed-out bar and an avg doesn't blow up. -export function isScoreShapedMetric(value: unknown): value is number { - return typeof value === 'number' && value >= 0 && value <= 1; +// Index of the max value, ignoring nulls; ties resolve to the first (left-most) +// entry, matching the version letter order users read. Returns null if every +// value is null. +export function indexOfMax(values: Array): number | null { + let best: number | null = null; + let bestValue = -Infinity; + values.forEach((value, index) => { + if (value !== null && value > bestValue) { + bestValue = value; + best = index; + } + }); + return best; } -// A run set is "running" while any run is still queued or executing, else -// "done". Shared by the collection card and the compare header so the two -// surfaces can't disagree; callers that also have a not-yet-loaded state keep -// their own `null` guard around this. +// Overall status of a run set: "running" while any run is still queued or +// executing, "error" when every run failed or was cancelled (so an all-failed +// collection doesn't read as a green "done"), otherwise "done". Shared by the +// collection card and the compare header so the two surfaces can't disagree; +// callers that also have a not-yet-loaded state keep their own `null` guard. export function deriveRunsStatus( runs: Array<{ status: EvalCollectionRunStatus }>, -): 'running' | 'done' { - return runs.some((run) => run.status === 'new' || run.status === 'running') ? 'running' : 'done'; +): 'running' | 'done' | 'error' { + if (runs.some((run) => run.status === 'new' || run.status === 'running')) return 'running'; + if ( + runs.length > 0 && + runs.every((run) => run.status === 'error' || run.status === 'cancelled') + ) { + return 'error'; + } + return 'done'; } -// Reduce per-run aggregate metrics to the score-shaped ([0, 1]) metrics that -// both the collection-card preview and the compare hero chart render. Returns -// one entry per metric (first-seen order) with a value per run aligned by -// index — `null` where a run lacks the metric, so a skipped metric never -// shifts later versions out of their color/letter slot. A metric is included -// only if it's score-shaped across every run that reported it, since the bar -// charts clamp to max=1 and an absolute count (tokens, latency) would render a -// meaningless maxed-out bar. +// Reduce per-run aggregate metrics to the score metrics that both the +// collection-card preview and the compare hero chart render, each normalized to +// [0, 1] by its scale (AI-judge metrics are 1–5 → /5; see `normalizeMetricScore`). +// Returns one entry per metric (first-seen order) with a value per run aligned by +// index — `null` where a run lacks the metric, so a skipped metric never shifts +// later versions out of their color/letter slot. A metric is included only if +// every run that reported it yields a score (operational counts like tokens and +// latency normalize to `null` and are dropped, since the bar charts clamp to +// max=1 and an absolute count would render a meaningless maxed-out bar). export function buildScoreShapedMetricGroups( runs: Array<{ metrics: Record | null }>, ): Array<{ key: string; values: Array }> { @@ -146,32 +162,45 @@ export function buildScoreShapedMetricGroups( const seen = new Set(); for (const run of runs) { for (const key of Object.keys(run.metrics ?? {})) { - // Skip predefined operational metrics (token counts, execution time) — - // they're absolute values, not scores, and would chart as a bogus - // percentage on the rare run where they land in [0, 1]. Matches the - // exclusion in `getUserDefinedMetricNames`. - if (PREDEFINED_METRIC_KEYS.has(key) || seen.has(key)) continue; + if (seen.has(key)) continue; seen.add(key); orderedKeys.push(key); } } - const scoreShapedKeys = orderedKeys.filter((key) => - runs.every((run) => { - const value = run.metrics?.[key]; - return value === undefined || isScoreShapedMetric(value); - }), + const scoreKeys = orderedKeys.filter( + (key) => + runs.some((run) => run.metrics?.[key] !== undefined) && + runs.every((run) => { + const value = run.metrics?.[key]; + return value === undefined || normalizeMetricScore(key, value) !== null; + }), ); - return scoreShapedKeys.map((key) => ({ + return scoreKeys.map((key) => ({ key, values: runs.map((run) => { const value = run.metrics?.[key]; - return typeof value === 'number' ? value : null; + return typeof value === 'number' ? normalizeMetricScore(key, value) : null; }), })); } +// Mean of a metrics map's score values, each normalized to [0, 1] by its scale. +// Returns null when no metric qualifies (only operational counts, or an empty +// map). Single definition so the cards, cases table, and hero chart can't +// disagree on what a case/run scored. +export function averageNormalizedScore( + metrics: Record | null | undefined, +): number | null { + if (!metrics) return null; + const values = Object.entries(metrics) + .map(([key, value]) => normalizeMetricScore(key, value)) + .filter((value): value is number => value !== null); + if (values.length === 0) return null; + return values.reduce((sum, value) => sum + value, 0) / values.length; +} + export function computeDelta( current: number | undefined, previous: number | undefined, @@ -259,10 +288,10 @@ export function formatMetricLabel(name: string): string { // `correctness` + `helpfulness` collapse into 'aiBased' (both LLM-as-judge). export function getMetricCategory(metric: string | undefined): MetricCategory { + if (metric !== undefined && (ONE_TO_FIVE_METRIC_KEYS as readonly string[]).includes(metric)) { + return 'aiBased'; + } switch (metric) { - case 'correctness': - case 'helpfulness': - return 'aiBased'; case 'stringSimilarity': return 'stringSimilarity'; case 'categorization': @@ -274,6 +303,23 @@ export function getMetricCategory(metric: string | undefined): MetricCategory { } } +// Short "what this measures" copy for the built-in metrics, mirrored from the +// Evaluation node's metric options. Custom/unknown metrics have no canned +// description (the UI just shows the name). +const METRIC_DESCRIPTION_KEYS: Partial> = { + correctness: 'evaluation.metric.description.correctness', + helpfulness: 'evaluation.metric.description.helpfulness', + stringSimilarity: 'evaluation.metric.description.stringSimilarity', + categorization: 'evaluation.metric.description.categorization', + toolsUsed: 'evaluation.metric.description.toolsUsed', +}; + +// i18n key for a metric's description, or null for custom/unknown metrics. +export function getMetricDescriptionKey(metric: string | undefined): BaseTextKey | null { + if (metric === undefined) return null; + return METRIC_DESCRIPTION_KEYS[metric] ?? null; +} + function formatScoreNumerator(value: number): string { const rounded = Math.round(value * 10) / 10; return Number.isInteger(rounded) ? `${rounded}` : rounded.toFixed(1); diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/CompareCollectionView.test.ts b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/CompareCollectionView.test.ts index af1f2e95cfe..c8e87df996d 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/CompareCollectionView.test.ts +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/CompareCollectionView.test.ts @@ -6,9 +6,15 @@ import { createComponentRenderer } from '@/__tests__/render'; import { VIEWS } from '@/app/constants'; import { useEvalCollectionsStore } from '../evalCollections.store'; +import { useEvaluationStore } from '../evaluation.store'; import type { EvaluationCollectionDetail } from '../evalCollections.types'; import CompareCollectionView from './CompareCollectionView.vue'; +const track = vi.fn(); +vi.mock('@/app/composables/useTelemetry', () => ({ + useTelemetry: () => ({ track }), +})); + const routerReplace = vi.fn(); vi.mock('vue-router', async (importOriginal) => ({ ...(await importOriginal()), @@ -82,18 +88,29 @@ const renderComponent = createComponentRenderer(CompareCollectionView, { describe('CompareCollectionView', () => { let store: ReturnType; + let evaluationStore: ReturnType; beforeEach(() => { flagState.enabled = true; routerReplace.mockClear(); + track.mockClear(); store = useEvalCollectionsStore(); + evaluationStore = useEvaluationStore(); store.stopPolling = vi.fn() as unknown as typeof store.stopPolling; + // Per-case fetch is stubbed to a no-op by default so useCompareCases + // doesn't hit the network; tests that assert on cases seed the map. + evaluationStore.fetchTestCaseExecutions = vi.fn( + async () => [], + ) as unknown as typeof evaluationStore.fetchTestCaseExecutions; // Hard-replace the shared testing pinia's maps so a prior test's cached // detail doesn't leak in (object-form `$patch` deep-merges stale keys). store.$patch((state) => { state.collectionDetailById = {}; state.loadingDetail = {}; }); + evaluationStore.$patch((state) => { + state.testCaseExecutionsById = {}; + }); }); it('redirects to the evaluations list when the flag is off', async () => { @@ -139,6 +156,33 @@ describe('CompareCollectionView', () => { expect(container.textContent).toContain('Tone tuning experiment'); }); + it('renders the compare tabs and fires the compare-opened event once data loads', async () => { + store.fetchCollectionDetail = vi.fn(async () => { + store.$patch({ collectionDetailById: { 'col-1': DETAIL } }); + return DETAIL; + }) as unknown as typeof store.fetchCollectionDetail; + + const { container } = renderComponent(); + + await waitFor(() => + expect(container.querySelector('[data-test-id="compare-tabs"]')).not.toBeNull(), + ); + await waitFor(() => + expect(track).toHaveBeenCalledWith( + 'Eval collection compared opened', + expect.objectContaining({ + workflow_id: 'wf-1', + collection_id: 'col-1', + version_count: 2, + }), + ), + ); + // fired exactly once for the collection + expect(track.mock.calls.filter((c) => c[0] === 'Eval collection compared opened')).toHaveLength( + 1, + ); + }); + it('shows the not-found state when the collection has no detail', async () => { store.fetchCollectionDetail = vi.fn( async () => undefined, diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/CompareCollectionView.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/CompareCollectionView.vue index d65e4af62f9..f0fad8c6ed3 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/CompareCollectionView.vue +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/CompareCollectionView.vue @@ -6,14 +6,19 @@ import { useRouter } from 'vue-router'; import { VIEWS } from '@/app/constants'; import { useToast } from '@/app/composables/useToast'; +import { useTelemetry } from '@/app/composables/useTelemetry'; import { usePostHog } from '@/app/stores/posthog.store'; import CompareHeader from '../components/Compare/CompareHeader.vue'; import ScoreChart from '../components/Compare/ScoreChart.vue'; import AiInsightsCard from '../components/Compare/AiInsightsCard.vue'; +import CompareTabs from '../components/Compare/CompareTabs.vue'; +import DatasetMismatchBanner from '../components/Compare/DatasetMismatchBanner.vue'; import { useCompareData } from '../composables/useCompareData'; +import { useCompareCases } from '../composables/useCompareCases'; import { useEvalCollectionsFlag } from '../composables/useEvalCollectionsFlag'; import { useEvalCollectionsStore } from '../evalCollections.store'; +import { useEvaluationStore } from '../evaluation.store'; const props = defineProps<{ workflowId: string; @@ -23,12 +28,60 @@ const props = defineProps<{ const i18n = useI18n(); const router = useRouter(); const toast = useToast(); +const telemetry = useTelemetry(); const store = useEvalCollectionsStore(); +const evaluationStore = useEvaluationStore(); const postHog = usePostHog(); const isEvalCollectionsEnabled = useEvalCollectionsFlag(); const detail = computed(() => store.getDetail(props.collectionId)); + +// metric name → its custom LLM-judge prompt (the specific criteria the user +// configured), sourced from the collection's evaluation config. Run metrics are +// keyed by the metric's `name` (see the workflow compiler), so the map keys line +// up with the compare view's metric keys. Empty until the config resolves. +const metricPrompts = computed>(() => { + const configId = detail.value?.evaluationConfigId; + if (!configId) return {}; + const config = (evaluationStore.evaluationConfigsByWorkflowId[props.workflowId] ?? []).find( + (candidate) => candidate.id === configId, + ); + if (!config) return {}; + const prompts: Record = {}; + for (const metric of config.metrics) { + if (metric.type === 'llm_judge' && metric.config.prompt) { + prompts[metric.name] = metric.config.prompt; + } + } + return prompts; +}); const { compareData } = useCompareData(detail); +const workflowIdRef = computed(() => props.workflowId); +const { + caseRows, + mismatch, + loading: casesLoading, + casesLoaded, + casesError, +} = useCompareCases(detail, workflowIdRef); + +// Fire the compare-opened event once per collection, after both the versions +// and the per-case data have resolved so `case_count` is accurate. +const tracked = ref(false); +watch( + () => compareData.value !== null && casesLoaded.value, + (ready) => { + if (!ready || tracked.value) return; + tracked.value = true; + telemetry.track('Eval collection compared opened', { + workflow_id: props.workflowId, + collection_id: props.collectionId, + version_count: compareData.value?.versions.length ?? 0, + case_count: mismatch.value.maxCount, + }); + }, + { immediate: true }, +); const loading = computed(() => store.loadingDetail[props.collectionId] ?? false); // Set only when the collection is genuinely gone (404), so a deleted collection @@ -49,13 +102,31 @@ function isNotFoundError(error: unknown): boolean { ); } +// Tracks unmount so a fetch that resolves after the user leaves can tear down +// the poll it armed instead of letting it outlive the view. +let unmounted = false; + async function load(workflowId: string, collectionId: string) { notFound.value = false; try { await store.fetchCollectionDetail(workflowId, collectionId); + // Best-effort: metric criteria come from the eval config. A failure here + // just means the compare view shows metric names without their criteria. + await evaluationStore.fetchEvaluationConfigs(workflowId).catch(() => null); + // If we left or switched collections mid-fetch, `fetchCollectionDetail` + // may have just (re)armed polling for a collection we're no longer + // showing — stop it so the timer doesn't outlive the view. + if (unmounted || collectionId !== props.collectionId) { + store.stopPolling(collectionId); + } } catch (error) { - toast.showError(error, i18n.baseText('evaluation.compare.errors.loadFailed')); - notFound.value = isNotFoundError(error); + // A 404 already shows the not-found state; a toast on top would be a + // second, contradictory signal. Only toast transient (non-404) failures. + if (isNotFoundError(error)) { + notFound.value = true; + } else { + toast.showError(error, i18n.baseText('evaluation.compare.errors.loadFailed')); + } } } @@ -84,11 +155,13 @@ watch( [() => props.workflowId, () => props.collectionId], ([, collectionId], [, prevCollectionId]) => { store.stopPolling(prevCollectionId); + tracked.value = false; void load(props.workflowId, collectionId); }, ); onBeforeUnmount(() => { + unmounted = true; store.stopPolling(props.collectionId); }); @@ -120,16 +193,33 @@ onBeforeUnmount(() => {
diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/EvalCollectionsListView.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/EvalCollectionsListView.vue index 30e4843b424..6669b9c2c96 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/EvalCollectionsListView.vue +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/EvalCollectionsListView.vue @@ -1,5 +1,5 @@ diff --git a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/EvaluationsView.vue b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/EvaluationsView.vue index 00493429667..bafb251c51a 100644 --- a/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/EvaluationsView.vue +++ b/packages/frontend/editor-ui/src/features/ai/evaluation.ee/views/EvaluationsView.vue @@ -17,6 +17,9 @@ import { N8nButton, N8nIcon, N8nPopover } from '@n8n/design-system'; const props = defineProps<{ workflowId: string; + // When set, the runs table paginates to this many rows per page (used when + // stacked alongside the collections list so both fit). Omitted → no paging. + runsPageSize?: number; }>(); const locale = useI18n(); @@ -260,6 +263,7 @@ watch(runningTestRun, (run) => { :class="$style.runs" :runs="runs" :workflow-id="props.workflowId" + :page-size="runsPageSize" /> diff --git a/packages/testing/playwright/pages/EvaluationComparePage.ts b/packages/testing/playwright/pages/EvaluationComparePage.ts new file mode 100644 index 00000000000..5ca72270e88 --- /dev/null +++ b/packages/testing/playwright/pages/EvaluationComparePage.ts @@ -0,0 +1,62 @@ +import type { Locator } from '@playwright/test'; + +import { BasePage } from './BasePage'; + +/** + * The multi-version eval-collection compare view + * (`/workflow/:workflowId/evaluation/collections/:collectionId/compare`). + */ +export class EvaluationComparePage extends BasePage { + async goto(workflowId: string, collectionId: string): Promise { + // The editor is an SPA — resolve on `domcontentloaded` rather than the full + // `load` event (which can lag on a cold route), then wait for the view. + await this.page.goto(`/workflow/${workflowId}/evaluation/collections/${collectionId}/compare`, { + waitUntil: 'domcontentloaded', + // Generous nav timeout: the dev server can cold-compile this route on + // first navigation. Harmless against the prebuilt editor in CI. + timeout: 60_000, + }); + await this.getView().waitFor({ state: 'visible', timeout: 30_000 }); + } + + getView(): Locator { + return this.page.getByTestId('compare-collection-view'); + } + + getHeader(): Locator { + return this.page.getByTestId('compare-header'); + } + + getScoreChart(): Locator { + return this.page.getByTestId('compare-score-chart'); + } + + getTabs(): Locator { + return this.page.getByTestId('compare-tabs'); + } + + getDatasetMismatchBanner(): Locator { + return this.page.getByTestId('compare-dataset-mismatch'); + } + + getCasesTable(): Locator { + return this.page.getByTestId('compare-cases-table'); + } + + getCaseRows(): Locator { + return this.page.getByTestId('compare-cases-row'); + } + + getOutputsTab(): Locator { + return this.page.getByTestId('compare-outputs-tab'); + } + + getOutputColumns(): Locator { + return this.page.getByTestId('compare-outputs-column'); + } + + /** Click a case row to drill into its side-by-side outputs. */ + async openCase(index: number): Promise { + await this.getCaseRows().nth(index).click(); + } +} diff --git a/packages/testing/playwright/pages/n8nPage.ts b/packages/testing/playwright/pages/n8nPage.ts index 2bdb1af1681..4913461d76f 100644 --- a/packages/testing/playwright/pages/n8nPage.ts +++ b/packages/testing/playwright/pages/n8nPage.ts @@ -22,6 +22,7 @@ import { CredentialsPage } from './CredentialsPage'; import { DataTableDetails } from './DataTableDetails'; import { DataTableView } from './DataTableView'; import { DemoPage } from './DemoPage'; +import { EvaluationComparePage } from './EvaluationComparePage'; import { ExecutionsPage } from './ExecutionsPage'; import { InstanceAiPage } from './InstanceAiPage'; import { KeycloakLoginPage } from './KeycloakLoginPage'; @@ -102,6 +103,7 @@ export class n8nPage { readonly workflows: WorkflowsPage; readonly notifications: NotificationsPage; readonly credentials: CredentialsPage; + readonly evaluationCompare: EvaluationComparePage; readonly executions: ExecutionsPage; readonly sideBar: SidebarPage; readonly dataTable: DataTableView; @@ -185,6 +187,7 @@ export class n8nPage { this.workflows = new WorkflowsPage(page); this.notifications = new NotificationsPage(page); this.credentials = new CredentialsPage(page); + this.evaluationCompare = new EvaluationComparePage(page); this.executions = new ExecutionsPage(page); this.sideBar = new SidebarPage(page); this.signIn = new SignInPage(page); diff --git a/packages/testing/playwright/tests/e2e/ai/eval-collections-compare.spec.ts b/packages/testing/playwright/tests/e2e/ai/eval-collections-compare.spec.ts new file mode 100644 index 00000000000..4f3819da811 --- /dev/null +++ b/packages/testing/playwright/tests/e2e/ai/eval-collections-compare.spec.ts @@ -0,0 +1,209 @@ +import { nanoid } from 'nanoid'; + +import { test, expect } from '../../../fixtures/base'; +import type { TestRequirements } from '../../../Types'; + +const COLLECTION_ID = 'col-e2e'; + +// Enable the eval-collections feature surface client-side; all eval REST calls +// are stubbed below, so no backend flag or real eval run is needed. +const requirements: TestRequirements = { + storage: { + N8N_EXPERIMENT_OVERRIDES: JSON.stringify({ '084_eval_collections': true }), + }, +}; + +function caseFor(runId: string, runIndex: number, score: number, question: string, output: string) { + return { + id: `${runId}-${runIndex}`, + testRunId: runId, + executionId: null, + status: 'success', + createdAt: '2026-01-01T00:00:00Z', + updatedAt: '2026-01-01T00:00:00Z', + runAt: '2026-01-01T00:00:00Z', + runIndex, + metrics: { helpfulness: score }, + inputs: { question }, + outputs: { output }, + }; +} + +// Per-run case executions. The "Capital of France?" case has the largest +// score spread (0.5 → 0.9), so the cases table — sorted by biggest regression +// first — puts it in the top row, which the drilldown assertion relies on. +const CASES: Record>> = { + 'run-a': [ + caseFor('run-a', 0, 0.5, 'Capital of France?', 'Paris'), + caseFor('run-a', 1, 0.8, 'What is 2+2?', '4'), + ], + 'run-b': [ + caseFor('run-b', 0, 0.9, 'Capital of France?', 'The capital of France is Paris.'), + caseFor('run-b', 1, 0.88, 'What is 2+2?', '2 + 2 equals 4.'), + ], +}; + +const json = (data: unknown) => ({ + contentType: 'application/json', + body: JSON.stringify({ data }), +}); + +test.describe( + 'Eval collection compare view @auth:owner', + { annotation: [{ type: 'owner', description: 'AI' }] }, + () => { + let workflowId: string; + + test.beforeEach(async ({ n8n, setupRequirements }) => { + await setupRequirements(requirements); + + const workflow = await n8n.api.workflows.createWorkflow({ + name: `Compare E2E ${nanoid()}`, + nodes: [], + connections: {}, + }); + workflowId = workflow.id; + + const record = { + id: COLLECTION_ID, + name: 'Tone tuning experiment', + description: null, + workflowId, + evaluationConfigId: 'cfg-1', + createdById: 'owner', + createdAt: '2026-01-01T00:00:00Z', + updatedAt: '2026-01-01T00:00:00Z', + runCount: 2, + }; + const detail = { + ...record, + runs: [ + { + testRunId: 'run-a', + workflowVersionId: 'v1', + status: 'completed', + runAt: '2026-01-01T00:00:00Z', + completedAt: '2026-01-01T00:05:00Z', + avgScore: 0.61, + metrics: { helpfulness: 0.61 }, + }, + { + testRunId: 'run-b', + workflowVersionId: 'v2', + status: 'completed', + runAt: '2026-01-01T00:00:00Z', + completedAt: '2026-01-01T00:06:00Z', + avgScore: 0.89, + metrics: { helpfulness: 0.89 }, + }, + ], + }; + const insights = { + generatedAt: '2026-01-01T00:07:00Z', + modelUsed: 'test', + status: 'ok', + insights: { + winner: { + versionLabel: 'B', + headline: 'B wins', + body: 'Higher helpfulness across cases.', + }, + regressions: [], + suggestedNext: { + headline: 'Try C', + body: 'Raise the temperature.', + hypothesis: 'More varied phrasing may help.', + }, + }, + }; + + // The evaluation root view only renders the compare route once the + // workflow has at least one test run (otherwise it shows the setup + // wizard), so stub the test-runs list too. + const testRuns = detail.runs.map((run) => ({ + id: run.testRunId, + workflowId, + status: 'completed', + metrics: run.metrics, + createdAt: '2026-01-01T00:00:00Z', + updatedAt: '2026-01-01T00:06:00Z', + runAt: run.runAt, + completedAt: run.completedAt, + collectionId: COLLECTION_ID, + })); + await n8n.page.route( + new RegExp(`/rest/workflows/${workflowId}/test-runs(?!/)`), + async (route) => await route.fulfill(json(testRuns)), + ); + + // Order doesn't matter — the globs are disjoint (`(?!/)` keeps the detail + // and list routes from swallowing the /insights and /runs sub-paths). + await n8n.page.route( + new RegExp(`/rest/workflows/${workflowId}/eval-collections/${COLLECTION_ID}/insights`), + async (route) => await route.fulfill(json(insights)), + ); + await n8n.page.route( + new RegExp(`/rest/workflows/${workflowId}/eval-collections/${COLLECTION_ID}(?!/)`), + async (route) => await route.fulfill(json(detail)), + ); + await n8n.page.route( + new RegExp(`/rest/workflows/${workflowId}/eval-collections(?!/)`), + async (route) => await route.fulfill(json([record])), + ); + await n8n.page.route( + new RegExp(`/rest/workflows/${workflowId}/test-runs/[^/]+/test-cases`), + async (route) => { + const runId = + route + .request() + .url() + .match(/test-runs\/([^/?]+)\/test-cases/)?.[1] ?? ''; + await route.fulfill(json(CASES[runId] ?? [])); + }, + ); + }); + + test('renders the hero and cases table, and drills into per-version outputs', async ({ + n8n, + }) => { + const compare = n8n.evaluationCompare; + await compare.goto(workflowId, COLLECTION_ID); + + await expect(compare.getHeader()).toContainText('Tone tuning experiment'); + await expect(compare.getScoreChart()).toBeVisible(); + await expect(compare.getTabs()).toBeVisible(); + // Cases tab (default) lists both seeded cases by their input. + await expect(compare.getCasesTable()).toContainText('Capital of France?'); + await expect(compare.getCasesTable()).toContainText('What is 2+2?'); + + // Drilling into a case row jumps to the side-by-side outputs, showing + // each version's distinct answer for that case. + await compare.openCase(0); + await expect(compare.getOutputsTab()).toBeVisible(); + await expect(compare.getOutputsTab()).toContainText('Paris'); + await expect(compare.getOutputsTab()).toContainText('The capital of France is Paris.'); + }); + + test('surfaces a dataset-mismatch banner when run case counts diverge', async ({ n8n }) => { + // Re-stub run-b with a single case so the counts diverge (2 vs 1). A + // later route registration takes precedence in Playwright. + await n8n.page.route( + new RegExp(`/rest/workflows/${workflowId}/test-runs/[^/]+/test-cases`), + async (route) => { + const runId = + route + .request() + .url() + .match(/test-runs\/([^/?]+)\/test-cases/)?.[1] ?? ''; + const cases = runId === 'run-b' ? CASES['run-b'].slice(0, 1) : (CASES[runId] ?? []); + await route.fulfill(json(cases)); + }, + ); + + const compare = n8n.evaluationCompare; + await compare.goto(workflowId, COLLECTION_ID); + + await expect(compare.getDatasetMismatchBanner()).toBeVisible(); + }); + }, +);