feat(editor): Add per-case compare tabs to eval collection compare view (#33880)

This commit is contained in:
Arvin A
2026-07-17 09:46:40 +00:00
committed by GitHub
parent e34343b4a0
commit dde77cb7d3
49 changed files with 3014 additions and 236 deletions
@@ -257,6 +257,13 @@ export interface FrontendSettings {
easyAIWorkflowOnboarded: boolean;
evaluation: {
quota: number;
/**
* Operator override (`N8N_EVAL_COLLECTIONS_ENABLED`) that force-enables the
* eval-collections surface. Surfaced here so the frontend gate works even
* when the in-browser PostHog client is disabled (telemetry off), where the
* `084_eval_collections` flag would otherwise never resolve.
*/
collectionsEnabled: boolean;
};
/** Backend modules that were initialized during startup. */
+3
View File
@@ -515,6 +515,9 @@ export {
export {
EVAL_COLLECTIONS_FLAG,
RESERVED_METRIC_KEYS,
ONE_TO_FIVE_METRIC_KEYS,
normalizeMetricScore,
evalCollectionVersionEntrySchema,
createEvaluationCollectionSchema,
CreateEvaluationCollectionDto,
@@ -0,0 +1,38 @@
import { normalizeMetricScore, RESERVED_METRIC_KEYS } from '../eval-collections.schema';
describe('normalizeMetricScore', () => {
it('excludes reserved operational metrics', () => {
for (const key of RESERVED_METRIC_KEYS) {
// Even a value that happens to land in [0, 1] is not a score.
expect(normalizeMetricScore(key, 0.5)).toBeNull();
expect(normalizeMetricScore(key, 1719)).toBeNull();
}
});
it('normalizes 1–5 AI-judge metrics onto [0, 1]', () => {
expect(normalizeMetricScore('correctness', 5)).toBe(1);
expect(normalizeMetricScore('correctness', 4)).toBe(0.8);
expect(normalizeMetricScore('correctness', 1)).toBe(0.2);
expect(normalizeMetricScore('helpfulness', 2.5)).toBe(0.5);
});
it('drops AI-judge values that fall outside the 1–5 range once scaled', () => {
// 6 / 5 = 1.2 → out of [0, 1]
expect(normalizeMetricScore('correctness', 6)).toBeNull();
expect(normalizeMetricScore('helpfulness', -1)).toBeNull();
});
it('passes through other metrics only when already in [0, 1]', () => {
expect(normalizeMetricScore('accuracy', 0.9)).toBe(0.9);
expect(normalizeMetricScore('accuracy', 0)).toBe(0);
expect(normalizeMetricScore('accuracy', 1)).toBe(1);
// Unknown-scale values outside [0, 1] can't be scaled → excluded.
expect(normalizeMetricScore('accuracy', 1.5)).toBeNull();
expect(normalizeMetricScore('accuracy', -0.2)).toBeNull();
});
it('returns null for NaN', () => {
expect(normalizeMetricScore('accuracy', Number.NaN)).toBeNull();
expect(normalizeMetricScore('correctness', Number.NaN)).toBeNull();
});
});
@@ -14,6 +14,45 @@ export type EvalCollectionRunStatus = 'new' | 'running' | 'completed' | 'error'
// available after `083_canvas_nodes_grouping`).
export const EVAL_COLLECTIONS_FLAG = '084_eval_collections';
// Metric keys every run emits automatically (token counts + execution time).
// They're absolute operational values, not quality scores, so the "avg score"
// derivation and the score-shaped charts exclude them and count only
// user-defined metrics. Single-sourced so FE and BE agree on what a "score"
// is (the FE builds its `PREDEFINED_METRIC_KEYS` set from this).
export const RESERVED_METRIC_KEYS = [
'promptTokens',
'completionTokens',
'totalTokens',
'executionTime',
] as const;
// User-defined metrics whose LLM-as-judge handlers return a 1–5 rating (rather
// than a 0–1 fraction). Every other user metric is assumed already normalized
// to [0, 1]. Single-sourced so FE and BE agree; the FE's `getMetricCategory`
// derives its `aiBased` category from this list.
export const ONE_TO_FIVE_METRIC_KEYS = ['correctness', 'helpfulness'] as const;
/**
* Normalize a raw metric value to a [0, 1] "score", or return null when the
* metric isn't a score we can chart/average:
* - reserved operational metrics (token counts, execution time) → null
* - 1–5 AI-judge metrics → value / 5
* - any other metric → passed through only if already in [0, 1]; an
* unknown-scale value outside that range can't be meaningfully scaled, so
* it's excluded rather than rendered as a bogus percentage.
*
* Shared by the FE compare surfaces and the BE avg-score/insights derivation
* so a "score" means the same thing everywhere.
*/
export function normalizeMetricScore(key: string, value: number): number | null {
if ((RESERVED_METRIC_KEYS as readonly string[]).includes(key)) return null;
if (Number.isNaN(value)) return null;
const normalized = (ONE_TO_FIVE_METRIC_KEYS as readonly string[]).includes(key)
? value / 5
: value;
return normalized >= 0 && normalized <= 1 ? normalized : null;
}
// Per-version entry on a create-collection request. Either reference an
// existing test run (`existingTestRunId`) to reuse it, or omit it and the
// service will schedule a fresh run pinned to `workflowVersionId`. A null
@@ -452,6 +452,48 @@ describe('EvaluationCollectionService', () => {
expect(b?.lastRun?.isCritical).toBe(true);
});
it('scores versions on normalized quality metrics, excluding token/time metrics', async () => {
const versions: WorkflowHistory[] = [
mock<WorkflowHistory>({
versionId: 'wfv-a',
name: 'A',
autosaved: false,
createdAt: new Date('2026-04-01'),
}),
mock<WorkflowHistory>({
versionId: 'wfv-b',
name: 'B',
autosaved: false,
createdAt: new Date('2026-04-02'),
}),
];
workflowHistoryRepo.find.mockResolvedValueOnce(versions);
// correctness is a 1–5 metric → /5; tokens/executionTime are operational
// and excluded, so avgScore is the correctness score alone rather than a
// token-dominated mean in the thousands.
testRunRepo.find.mockResolvedValueOnce([
makeTestRun({
id: 'tr-a',
workflowVersionId: 'wfv-a',
metrics: { correctness: 5, completionTokens: 1719, executionTime: 21368 },
}),
makeTestRun({
id: 'tr-b',
workflowVersionId: 'wfv-b',
metrics: { correctness: 2, completionTokens: 540, executionTime: 11646 },
}),
]);
const result = await service.getEvalVersions('wf-1', 'cfg-1');
const a = result.versions.find((v) => v.workflowVersionId === 'wfv-a');
const b = result.versions.find((v) => v.workflowVersionId === 'wfv-b');
expect(a?.lastRun?.avgScore).toBe(1); // 5 / 5
expect(b?.lastRun?.avgScore).toBeCloseTo(0.4); // 2 / 5
expect(a?.lastRun?.isBest).toBe(true);
expect(b?.lastRun?.isCritical).toBe(true); // 0.4 < 0.6
});
it('includes a current-draft row with no last run', async () => {
workflowHistoryRepo.find.mockResolvedValueOnce([]);
const result = await service.getEvalVersions('wf-1', 'cfg-1');
@@ -7,6 +7,7 @@ import type {
EvaluationCollectionRunSummary,
UpdateEvaluationCollectionPayload,
} from '@n8n/api-types';
import { normalizeMetricScore } from '@n8n/api-types';
import type { TestRun, User } from '@n8n/db';
import {
EvaluationCollectionRepository,
@@ -179,6 +180,11 @@ export class EvaluationCollectionService {
workflowVersionId: versionId,
evaluationConfigId: input.evaluationConfigId,
evaluationConfigSnapshot: configSnapshot,
// Compile the eval config (dataset + trigger + metric nodes) onto
// each version's snapshot. Without this the run goes "direct" and
// the raw versioned workflow has no evaluation trigger → the run
// fails immediately with EVALUATION_TRIGGER_NOT_FOUND.
compileFromConfig: true,
},
);
runsStartedIds.push(testRun.id);
@@ -474,10 +480,16 @@ export class EvaluationCollectionService {
private computeAvgScore(run: TestRun): number | null {
const coerced = this.coerceMetrics(run.metrics);
if (!coerced) return null;
const values = Object.values(coerced);
if (values.length === 0) return null;
const sum = values.reduce((acc, v) => acc + v, 0);
return sum / values.length;
// A "score" is a user-defined metric normalized to [0, 1] by its scale
// (AI-judge metrics are 1–5 → /5). Operational metrics (token counts,
// execution time) normalize to null and are excluded. Mirrors the FE's
// score model so the versions table's %/best/critical annotations match
// the compare view; null when a run reports no score metric.
const scores = Object.entries(coerced)
.map(([key, value]) => normalizeMetricScore(key, value))
.filter((value): value is number => value !== null);
if (scores.length === 0) return null;
return scores.reduce((acc, value) => acc + value, 0) / scores.length;
}
private coerceMetrics(
@@ -218,6 +218,36 @@ describe('EvalInsightsService', () => {
expect(fluencyRegression).toBeUndefined();
});
it('scores on normalized quality metrics, ignoring token/time operational metrics', async () => {
collectionRepo.getDetailByIdAndWorkflowId.mockResolvedValueOnce({
collection: makeCollection(),
runs: [
// A wins on correctness (5/5) despite far more tokens/time; B is
// worse on correctness (2/5). The winner + the only regression must
// be about correctness — never completionTokens / executionTime.
makeRun({
id: 'tr-a',
metrics: { correctness: 5, completionTokens: 1719, executionTime: 21368 },
}),
makeRun({
id: 'tr-b',
metrics: { correctness: 2, completionTokens: 540, executionTime: 11646 },
}),
],
});
const result = await service.generateInsights(user, 'wf-1', 'col-1');
expect(result.insights.winner.versionLabel).toBe('A');
// correctness 5/5 → 100%, counted as a single score metric
expect(result.insights.winner.body).toContain('100%');
expect(result.insights.winner.body).toContain('1 metric');
const regressedMetrics = result.insights.regressions.map((r) => r.metric);
expect(regressedMetrics).toContain('correctness');
expect(regressedMetrics).not.toContain('completionTokens');
expect(regressedMetrics).not.toContain('executionTime');
});
it('returns a "lock in baseline" recommendation when no regressions cross the threshold', async () => {
collectionRepo.getDetailByIdAndWorkflowId.mockResolvedValueOnce({
collection: makeCollection(),
@@ -1,5 +1,5 @@
import type { AiInsightsPayload, AiInsightsResponse } from '@n8n/api-types';
import { aiInsightsResponseSchema } from '@n8n/api-types';
import { aiInsightsResponseSchema, normalizeMetricScore } from '@n8n/api-types';
import { LicenseState, Logger } from '@n8n/backend-common';
import type { TestRun, User } from '@n8n/db';
import { EvaluationCollectionRepository } from '@n8n/db';
@@ -27,7 +27,10 @@ type RunSummary = {
versionLabel: string;
workflowVersionId: string | null;
avgScore: number | null;
metrics: Record<string, number>;
// Per-metric scores normalized to [0, 1] (operational metrics excluded).
// Winner/regression comparisons run over these, not the raw metrics, so the
// insights talk about actual quality scores rather than token totals.
scores: Record<string, number>;
};
/**
@@ -156,33 +159,31 @@ export class EvalInsightsService {
// ---- internals ----
private summariseRun(run: TestRun, index: number): RunSummary {
const metrics = this.coerceMetrics(run.metrics);
const avg = this.averageScore(metrics);
const scores = this.scoreMetrics(run.metrics);
const values = Object.values(scores);
const avgScore =
values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : null;
// Label by index letter — A/B/C — so the FE legend chips line up. The
// agent (and the deterministic path) refer back to these labels.
const versionLabel = String.fromCharCode(0x41 + index);
return {
versionLabel,
workflowVersionId: run.workflowVersionId,
avgScore: avg,
metrics,
};
return { versionLabel, workflowVersionId: run.workflowVersionId, avgScore, scores };
}
private averageScore(metrics: Record<string, number>): number | null {
const values = Object.values(metrics);
if (values.length === 0) return null;
return values.reduce((sum, v) => sum + v, 0) / values.length;
}
private coerceMetrics(
// Per-metric scores normalized to [0, 1] by their scale (AI-judge metrics
// are 1–5 → /5); operational metrics (token counts, execution time) and
// unknown-scale values are dropped. Booleans coerce to 0/1. Shares the
// `normalizeMetricScore` contract with the FE + versions-table scoring so
// winner/regressions reflect the same "score" the user sees.
private scoreMetrics(
metrics: Record<string, number | boolean> | null | undefined,
): Record<string, number> {
if (!metrics) return {};
const out: Record<string, number> = {};
for (const [k, v] of Object.entries(metrics)) {
if (typeof v === 'number') out[k] = v;
else if (typeof v === 'boolean') out[k] = v ? 1 : 0;
for (const [key, raw] of Object.entries(metrics)) {
const value = typeof raw === 'boolean' ? (raw ? 1 : 0) : raw;
if (typeof value !== 'number') continue;
const score = normalizeMetricScore(key, value);
if (score !== null) out[key] = score;
}
return out;
}
@@ -252,7 +253,7 @@ export class EvalInsightsService {
winner: {
versionLabel: winner.versionLabel,
headline: `${winner.versionLabel} is the winner`,
body: `${winner.versionLabel} leads on average score (${this.formatScore(winner.avgScore)}) across ${Object.keys(winner.metrics).length} metric(s).`,
body: `${winner.versionLabel} leads on average score (${this.formatScore(winner.avgScore)}) across ${Object.keys(winner.scores).length} metric(s).`,
},
regressions,
suggestedNext,
@@ -269,17 +270,20 @@ export class EvalInsightsService {
const regressions: AiInsightsPayload['regressions'] = [];
for (const run of scored) {
if (run.versionLabel === winner.versionLabel) continue;
for (const [metric, winnerScore] of Object.entries(winner.metrics)) {
const runScore = run.metrics[metric];
for (const [metric, winnerScore] of Object.entries(winner.scores)) {
const runScore = run.scores[metric];
if (typeof runScore !== 'number') continue;
const delta = runScore - winnerScore;
if (delta >= -REGRESSION_DELTA_THRESHOLD) continue;
// `delta` is a [-1, 1] score difference; report it as signed
// percentage points so the copy reads "N percentage points below".
const deltaPoints = Number((delta * 100).toFixed(1));
regressions.push({
versionLabel: run.versionLabel,
metric,
delta: Number((delta * 100).toFixed(1)),
delta: deltaPoints,
headline: `${run.versionLabel} regressed on ${metric}`,
body: `${run.versionLabel} scored ${this.formatScore(runScore)} on ${metric}, ${this.formatScore(Math.abs(delta * 100))} percentage points below ${winner.versionLabel}.`,
body: `${run.versionLabel} scored ${this.formatScore(runScore)} on ${metric}, ${Math.abs(deltaPoints).toFixed(1)} percentage points below ${winner.versionLabel}.`,
});
}
}
@@ -306,7 +310,8 @@ export class EvalInsightsService {
}
private formatScore(score: number): string {
// Compact 0–1 score formatting; agent prose can override per locale.
return score.toFixed(2);
// Scores are normalized to [0, 1]; surface them as whole-percent so the
// prose reads "83%" rather than "0.83". Agent prose can override per locale.
return `${Math.round(score * 100)}%`;
}
}
@@ -2706,6 +2706,49 @@ describe('TestRunnerService', () => {
);
});
test('compiles from the supplied snapshot and does not re-read the live config', async () => {
// Collection runs pass a frozen config snapshot so a config edit racing
// the run can't change later versions' compilation. The runner must
// compile from that snapshot and skip the live DB lookup.
evaluationConfigRepository.findByIdAndWorkflowId.mockClear();
workflowRepository.findById.mockResolvedValueOnce({
id: 'wf-1',
name: 'Live',
nodes: [{ name: 'LiveNode' }],
connections: {},
settings: {},
} as never);
workflowHistoryService.findVersion.mockResolvedValueOnce({
versionId: 'wfv-x',
nodes: [{ name: 'SnapshotNode' } as never],
connections: {} as never,
} as never);
testRunRepository.createTestRun.mockResolvedValueOnce(mock<TestRun>({ id: 'tr-snap' }));
const snapshot = { id: 'cfg-1', name: 'frozen', metrics: [] } as never;
let compiledConfig: unknown;
workflowCompiler.compile.mockImplementationOnce((wf, config) => {
compiledConfig = config;
return wf as never;
});
const { finished } = await testRunnerService.startTestRun(USER as never, 'wf-1', 1, {
collectionId: 'col-1',
workflowVersionId: 'wfv-x',
evaluationConfigId: 'cfg-1',
evaluationConfigSnapshot: snapshot,
compileFromConfig: true,
});
await finished.catch(() => undefined);
expect(evaluationConfigRepository.findByIdAndWorkflowId).not.toHaveBeenCalled();
expect(compiledConfig).toBe(snapshot);
expect(testRunRepository.createTestRun).toHaveBeenCalledWith(
'wf-1',
expect.objectContaining({ evaluationConfigSnapshot: snapshot }),
);
});
test('overwrites workflowData.versionId with the pinned history versionId', async () => {
// `ExecutionPersistence` reads `workflowData.versionId` and stores
// it on the execution row. If the live workflow's `versionId` leaks
@@ -620,16 +620,25 @@ export class TestRunnerService {
let evaluationConfigSnapshot = options?.evaluationConfigSnapshot ?? null;
let configToCompile: EvaluationConfig | undefined;
let configLookupErrorCode: typeof TestRunErrorCode.EVALUATION_CONFIG_NOT_FOUND | undefined;
if (options?.compileFromConfig && options?.evaluationConfigId) {
const config = await this.evaluationConfigRepository.findByIdAndWorkflowId(
options.evaluationConfigId,
workflowId,
);
if (!config) {
configLookupErrorCode = TestRunErrorCode.EVALUATION_CONFIG_NOT_FOUND;
} else {
configToCompile = config;
evaluationConfigSnapshot = config as unknown as IDataObject;
if (options?.compileFromConfig) {
if (options.evaluationConfigSnapshot) {
// Prefer the caller-supplied frozen snapshot: collection runs pass one
// so every version compiles against identical dataset/trigger/metrics,
// immune to a config edit that races the run. Its serialized dates are
// not read by the compiler, which only needs the eval fields.
configToCompile = options.evaluationConfigSnapshot as unknown as EvaluationConfig;
} else if (options.evaluationConfigId) {
// No snapshot supplied (single-run callers) — resolve the live config.
const config = await this.evaluationConfigRepository.findByIdAndWorkflowId(
options.evaluationConfigId,
workflowId,
);
if (!config) {
configLookupErrorCode = TestRunErrorCode.EVALUATION_CONFIG_NOT_FOUND;
} else {
configToCompile = config;
evaluationConfigSnapshot = config as unknown as IDataObject;
}
}
}
@@ -419,6 +419,7 @@ export class FrontendService {
},
evaluation: {
quota: this.licenseState.getMaxWorkflowsWithEvaluations(),
collectionsEnabled: this.globalConfig.evaluation.collectionsEnabled,
},
activeModules: this.moduleRegistry.getActiveModules(),
canvasOnly: this.globalConfig.canvasOnly,
@@ -5721,6 +5721,7 @@
"evaluation.collections.errors.fetchFailed": "Could not load evaluation collections",
"evaluation.collections.card.done": "Done",
"evaluation.collections.card.running": "Running",
"evaluation.collections.card.failed": "Failed",
"evaluation.collections.card.currentDraft": "Current draft",
"evaluation.collections.card.meta.versions": "{count} version | {count} versions",
"evaluation.collections.card.lastRunToday": "today, {time}",
@@ -5756,6 +5757,37 @@
"evaluation.compare.insights.noRegressions": "No regressions detected",
"evaluation.compare.errors.loadFailed": "Could not load the comparison.",
"evaluation.compare.errors.notFound": "This collection could not be found.",
"evaluation.compare.tabs.cases": "Cases",
"evaluation.compare.tabs.outputs": "Outputs",
"evaluation.compare.tabs.metrics": "Metrics",
"evaluation.compare.tabs.workflowDiff": "Workflow diff",
"evaluation.compare.datasetMismatch": "Versions ran against different case counts ({counts}). Comparison is aligned by row index.",
"evaluation.compare.cases.col.index": "#",
"evaluation.compare.cases.col.input": "Case input",
"evaluation.compare.cases.col.best": "Best",
"evaluation.compare.cases.col.deltaVsBest": "Δ vs best",
"evaluation.compare.cases.bestPill": "{letter} ★",
"evaluation.compare.cases.loading": "Loading cases…",
"evaluation.compare.cases.running": "Evaluation in progress — scores fill in as each case completes.",
"evaluation.compare.cases.empty": "No case-level results to compare yet.",
"evaluation.compare.cases.loadError": "Some versions' case data couldn't be loaded.",
"evaluation.compare.outputs.casesSidebarTitle": "Test cases",
"evaluation.compare.outputs.input": "Input",
"evaluation.compare.outputs.noOutput": "No output for this version.",
"evaluation.compare.metrics.col.metric": "Metric",
"evaluation.compare.metrics.empty": "No score-shaped metrics to compare yet.",
"evaluation.metric.description.correctness": "Whether the answer’s meaning matches a reference answer. Scored 1 (worst)–5 (best).",
"evaluation.metric.description.helpfulness": "Whether the response addresses the query. Scored 1 (worst)–5 (best).",
"evaluation.metric.description.stringSimilarity": "How close the answer is to a reference answer, character by character. Scored 0–1.",
"evaluation.metric.description.categorization": "Whether the answer exactly matches the reference answer. 1 if so, 0 otherwise.",
"evaluation.metric.description.toolsUsed": "Whether the expected tool(s) were used. Scored 0–1.",
"evaluation.metric.criteria.label": "Criteria:",
"evaluation.metric.criteria.showMore": "Show more",
"evaluation.metric.criteria.showLess": "Show less",
"evaluation.compare.workflowDiff.needTwo": "Add at least two versions to this collection to compare their workflows.",
"evaluation.compare.workflowDiff.loadError": "Could not load the workflows to compare.",
"evaluation.compare.workflowDiff.base": "Base",
"evaluation.compare.workflowDiff.compare": "Compare",
"evaluation.setup.title": "New collection",
"evaluation.setup.subtitle": "Pick a dataset and the versions you want to compare.",
"evaluation.setup.collectionName": "Collection name",
@@ -183,6 +183,7 @@ export const defaultSettings: FrontendSettings = {
},
evaluation: {
quota: 0,
collectionsEnabled: false,
},
activeModules: [],
canvasOnly: false,
@@ -0,0 +1,123 @@
import { fireEvent } from '@testing-library/vue';
import { describe, expect, it } from 'vitest';
import { createComponentRenderer } from '@/__tests__/render';
import type { CompareCaseCell, CompareCaseRow } from '../../composables/useCompareCases';
import type { CompareVersion } from '../../composables/useCompareData';
import CasesTable from './CasesTable.vue';
const versions: CompareVersion[] = [
{
index: 0,
testRunId: 'run-a',
workflowVersionId: 'v0',
letter: 'A',
label: 'v0',
status: 'completed',
avgScore: null,
},
{
index: 1,
testRunId: 'run-b',
workflowVersionId: 'v1',
letter: 'B',
label: 'v1',
status: 'completed',
avgScore: null,
},
];
// A null score with no `testCaseId` represents a case this version's run never
// covered (dataset drift → `⊘`); a null score with a `testCaseId` is a case
// still running (→ `–`).
const cell = (versionIndex: number, score: number | null): CompareCaseCell => ({
versionIndex,
testCaseId: score === null ? null : `c${versionIndex}`,
inputs: { q: 'x' },
outputs: { output: 'y' },
metrics: score === null ? undefined : { helpfulness: score },
score,
});
const row = (index: number, scores: Array<number | null>, best: number | null): CompareCaseRow => ({
index,
displayIndex: index + 1,
inputPreview: `case ${index}`,
cells: scores.map((s, i) => cell(i, s)),
bestVersionIndex: best,
});
const renderComponent = createComponentRenderer(CasesTable);
describe('CasesTable', () => {
it('renders one row per case with the best-version pill', () => {
const { container } = renderComponent({
props: { versions, caseRows: [row(0, [0.7, 0.9], 1), row(1, [0.6, 0.5], 0)] },
});
expect(container.querySelectorAll('[data-test-id="compare-cases-row"]')).toHaveLength(2);
// best pills render the winning version's letter
expect(container.textContent).toContain('B ★');
expect(container.textContent).toContain('A ★');
});
it('sorts by score spread descending by default (biggest regression first)', () => {
const { container } = renderComponent({
props: {
versions,
// row 0 spread 0.05, row 1 spread 0.4 → row 1 should come first
caseRows: [row(0, [0.9, 0.85], 0), row(1, [0.9, 0.5], 0)],
},
});
const rows = container.querySelectorAll('[data-test-id="compare-cases-row"]');
// first rendered row is the bigger-spread case (#2)
expect(rows[0].textContent).toContain('2');
});
it('emits drilldown with the case index when a row is clicked', async () => {
const { container, emitted } = renderComponent({
props: { versions, caseRows: [row(3, [0.7, 0.9], 1)] },
});
await fireEvent.click(container.querySelector('[data-test-id="compare-cases-row"]')!);
expect(emitted().drilldown).toEqual([[3]]);
});
it('renders a missing-case marker (⊘) for a case absent from a version', () => {
const { container } = renderComponent({
props: { versions, caseRows: [row(0, [0.7, null], 0)] },
});
expect(container.textContent).toContain('⊘');
});
it('renders a pending marker (–) for a scored-yet case that still exists', () => {
const pendingCell: CompareCaseCell = {
versionIndex: 1,
testCaseId: 'c1',
inputs: {},
outputs: undefined,
metrics: undefined,
score: null,
};
const { container } = renderComponent({
props: {
versions,
caseRows: [
{
index: 0,
displayIndex: 1,
inputPreview: 'case 0',
cells: [cell(0, 0.7), pendingCell],
bestVersionIndex: 0,
},
],
},
});
expect(container.textContent).toContain('–');
expect(container.textContent).not.toContain('⊘');
});
});
@@ -0,0 +1,246 @@
<script setup lang="ts">
import { N8nIcon, N8nText } from '@n8n/design-system';
import { useI18n } from '@n8n/i18n';
import { computed, ref } from 'vue';
import type { CompareCaseCell, CompareCaseRow } from '../../composables/useCompareCases';
import type { CompareVersion } from '../../composables/useCompareData';
import { computeDelta, formatDeltaPercent, formatMetricPercent } from '../../evaluation.utils';
import { versionColorVar } from '../shared/versionPalette';
const props = defineProps<{
versions: CompareVersion[];
caseRows: CompareCaseRow[];
}>();
const emit = defineEmits<{
drilldown: [caseIndex: number];
}>();
const i18n = useI18n();
// Below this fraction a score reads as a critical regression (mirrors the hero
// chart threshold).
const CRITICAL_THRESHOLD = 0.6;
type SortKey = 'spread' | 'index' | 'best';
// Default to the biggest regressions first, so the cases that moved most surface
// at the top without scrolling. "spread" = the gap between a case's best and
// worst version, which is what the "Δ vs best" header sorts on.
const sort = ref<{ key: SortKey; dir: 'asc' | 'desc' }>({ key: 'spread', dir: 'desc' });
function toggleSort(key: SortKey) {
if (sort.value.key === key) {
sort.value = { key, dir: sort.value.dir === 'asc' ? 'desc' : 'asc' };
} else {
sort.value = { key, dir: key === 'index' ? 'asc' : 'desc' };
}
}
function bestScore(row: CompareCaseRow): number | null {
if (row.bestVersionIndex === null) return null;
return row.cells[row.bestVersionIndex].score;
}
// Spread between the best and worst scored version — the "regression size" the
// default sort ranks by.
function scoreSpread(row: CompareCaseRow): number {
const scores = row.cells.map((cell) => cell.score).filter((s): s is number => s !== null);
if (scores.length < 2) return 0;
return Math.max(...scores) - Math.min(...scores);
}
const sortedRows = computed(() => {
const rows = [...props.caseRows];
const dir = sort.value.dir === 'asc' ? 1 : -1;
const value = (row: CompareCaseRow) => {
if (sort.value.key === 'index') return row.index;
if (sort.value.key === 'best') return bestScore(row) ?? -1;
return scoreSpread(row);
};
return rows.sort((a, b) => (value(a) - value(b)) * dir);
});
function isCritical(score: number | null): boolean {
return score !== null && score < CRITICAL_THRESHOLD;
}
// A scored cell shows its percentage; an unscored one is either still running
// (the case exists — `testCaseId` set — but hasn't produced a metric yet, shown
// as a neutral dash) or genuinely absent from this version's run (dataset
// drift → `⊘`).
function scoreLabel(cell: CompareCaseCell): string {
if (cell.score !== null) return formatMetricPercent(cell.score);
return cell.testCaseId !== null ? '–' : '⊘';
}
// Non-winning versions' delta vs the best, for the "Δ vs best" column.
function deltas(row: CompareCaseRow) {
const best = bestScore(row);
if (best === null || row.bestVersionIndex === null) return [];
return row.cells
.filter((cell) => cell.versionIndex !== row.bestVersionIndex && cell.score !== null)
.map((cell) => ({
versionIndex: cell.versionIndex,
letter: props.versions[cell.versionIndex]?.letter ?? '',
delta: formatDeltaPercent(computeDelta(cell.score ?? undefined, best)),
}));
}
</script>
<template>
<table :class="$style.table" data-test-id="compare-cases-table">
<thead>
<tr>
<th :class="$style.num" role="button" @click="toggleSort('index')">
{{ i18n.baseText('evaluation.compare.cases.col.index') }}
</th>
<th>{{ i18n.baseText('evaluation.compare.cases.col.input') }}</th>
<th v-for="version in versions" :key="version.testRunId" :class="$style.score">
{{ version.letter }}
</th>
<th role="button" @click="toggleSort('best')">
{{ i18n.baseText('evaluation.compare.cases.col.best') }}
</th>
<th role="button" @click="toggleSort('spread')">
{{ i18n.baseText('evaluation.compare.cases.col.deltaVsBest') }}
</th>
<th :class="$style.chevronCol" />
</tr>
</thead>
<tbody>
<tr
v-for="row in sortedRows"
:key="row.index"
:class="$style.row"
tabindex="0"
data-test-id="compare-cases-row"
@click="emit('drilldown', row.index)"
@keydown.enter="emit('drilldown', row.index)"
>
<td :class="$style.num">{{ row.displayIndex }}</td>
<td :class="$style.input" :title="row.inputPreview">{{ row.inputPreview }}</td>
<td v-for="cell in row.cells" :key="cell.versionIndex" :class="$style.score">
<span :class="$style.chip">
<span :class="$style.dot" :style="{ background: versionColorVar(cell.versionIndex) }" />
<N8nText size="xsmall" :color="isCritical(cell.score) ? 'danger' : 'text-base'" bold>
{{ scoreLabel(cell) }}
</N8nText>
</span>
</td>
<td>
<span v-if="row.bestVersionIndex !== null" :class="$style.bestPill">
<N8nText size="xsmall" color="success" bold>
{{
i18n.baseText('evaluation.compare.cases.bestPill', {
interpolate: { letter: versions[row.bestVersionIndex]?.letter ?? '' },
})
}}
</N8nText>
</span>
</td>
<td>
<span :class="$style.deltas">
<N8nText
v-for="d in deltas(row)"
:key="d.versionIndex"
size="xsmall"
color="text-light"
>
{{ d.letter }} {{ d.delta }}
</N8nText>
</span>
</td>
<td :class="$style.chevronCol">
<N8nIcon icon="chevron-right" size="small" color="text-light" />
</td>
</tr>
</tbody>
</table>
</template>
<style module lang="scss">
.table {
width: 100%;
border-collapse: collapse;
font-size: var(--font-size--xs);
th {
text-align: left;
padding: var(--spacing--2xs) var(--spacing--xs);
color: var(--text-color--subtle);
font-weight: var(--font-weight--medium);
border-bottom: 1px solid var(--border-color--subtle);
white-space: nowrap;
&[role='button'] {
cursor: pointer;
user-select: none;
}
}
td {
padding: var(--spacing--2xs) var(--spacing--xs);
border-bottom: 1px solid var(--border-color--subtle);
vertical-align: middle;
}
}
.row {
cursor: pointer;
&:hover {
background: var(--background--subtle);
}
}
.num {
width: 40px;
color: var(--text-color--subtle);
}
.input {
max-width: 320px;
overflow: hidden;
text-overflow: ellipsis;
white-space: nowrap;
}
.score {
text-align: center;
white-space: nowrap;
}
.chip {
display: inline-flex;
align-items: center;
gap: var(--spacing--4xs);
}
.dot {
width: 8px;
height: 8px;
border-radius: var(--radius--full);
flex-shrink: 0;
}
.bestPill {
display: inline-flex;
align-items: center;
padding: var(--spacing--5xs) var(--spacing--2xs);
border-radius: var(--radius--full);
border: 1px solid var(--border-color--success);
white-space: nowrap;
}
.deltas {
display: inline-flex;
gap: var(--spacing--2xs);
white-space: nowrap;
}
.chevronCol {
width: 32px;
text-align: right;
}
</style>
@@ -56,4 +56,20 @@ describe('CompareHeader', () => {
expect(container.textContent).toContain('Running');
});
it('shows the failed badge when every run errored', () => {
const { container } = renderComponent({
props: {
collectionName: 'Exp',
versions: [
version({ index: 0, status: 'error', avgScore: null }),
version({ index: 1, status: 'cancelled', avgScore: null }),
],
bestVersionIndex: null,
},
});
expect(container.textContent).toContain('Failed');
expect(container.textContent).not.toContain('Done');
});
});
@@ -17,6 +17,26 @@ const i18n = useI18n();
const status = computed(() => deriveRunsStatus(props.versions));
const statusBadge = computed(() => {
switch (status.value) {
case 'error':
return {
theme: 'warning' as const,
label: i18n.baseText('evaluation.collections.card.failed'),
};
case 'running':
return {
theme: 'tertiary' as const,
label: i18n.baseText('evaluation.collections.card.running'),
};
default:
return {
theme: 'success' as const,
label: i18n.baseText('evaluation.collections.card.done'),
};
}
});
const legend = computed(() =>
props.versions.map((version) => ({
...version,
@@ -30,15 +50,7 @@ const legend = computed(() =>
<header :class="$style.header" data-test-id="compare-header">
<div :class="$style.titleRow">
<N8nText tag="h2" size="xlarge" bold>{{ collectionName }}</N8nText>
<N8nBadge :theme="status === 'done' ? 'success' : 'tertiary'" size="small">
{{
i18n.baseText(
status === 'done'
? 'evaluation.collections.card.done'
: 'evaluation.collections.card.running',
)
}}
</N8nBadge>
<N8nBadge :theme="statusBadge.theme" size="small">{{ statusBadge.label }}</N8nBadge>
</div>
<N8nText size="small" color="text-light">
{{
@@ -0,0 +1,92 @@
import { fireEvent, waitFor } from '@testing-library/vue';
import { describe, expect, it } from 'vitest';
import { createComponentRenderer } from '@/__tests__/render';
import type { CompareCaseRow } from '../../composables/useCompareCases';
import type { CompareMetricGroup, CompareVersion } from '../../composables/useCompareData';
import CompareTabs from './CompareTabs.vue';
const versions: CompareVersion[] = [
{
index: 0,
testRunId: 'run-a',
workflowVersionId: 'v0',
letter: 'A',
label: 'v0',
status: 'completed',
avgScore: null,
},
{
index: 1,
testRunId: 'run-b',
workflowVersionId: 'v1',
letter: 'B',
label: 'v1',
status: 'completed',
avgScore: null,
},
];
const metricGroups: CompareMetricGroup[] = [
{ key: 'helpfulness', label: 'Helpfulness', values: [0.7, 0.9], bestIndex: 1 },
];
const caseRows: CompareCaseRow[] = [
{
index: 0,
displayIndex: 1,
inputPreview: 'case 0',
cells: [
{
versionIndex: 0,
testCaseId: 'a',
inputs: {},
outputs: { output: 'x' },
metrics: { helpfulness: 0.7 },
score: 0.7,
},
{
versionIndex: 1,
testCaseId: 'b',
inputs: {},
outputs: { output: 'y' },
metrics: { helpfulness: 0.9 },
score: 0.9,
},
],
bestVersionIndex: 1,
},
];
const renderComponent = createComponentRenderer(CompareTabs);
describe('CompareTabs', () => {
it('shows the Cases tab by default', () => {
const { container } = renderComponent({
props: { versions, metricGroups, caseRows, casesLoading: false },
});
expect(container.querySelector('[data-test-id="compare-cases-table"]')).not.toBeNull();
});
it('shows a loading message while cases load', () => {
const { container } = renderComponent({
props: { versions, metricGroups, caseRows: [], casesLoading: true },
});
expect(container.querySelector('[data-test-id="compare-cases-table"]')).toBeNull();
expect(container.textContent).toContain('Loading cases');
});
it('drilling into a case row switches to the Outputs tab', async () => {
const { container } = renderComponent({
props: { versions, metricGroups, caseRows, casesLoading: false },
});
await fireEvent.click(container.querySelector('[data-test-id="compare-cases-row"]')!);
await waitFor(() =>
expect(container.querySelector('[data-test-id="compare-outputs-tab"]')).not.toBeNull(),
);
});
});
@@ -0,0 +1,127 @@
<script setup lang="ts">
import { N8nTabs, N8nText } from '@n8n/design-system';
import type { TabOptions } from '@n8n/design-system';
import { useI18n } from '@n8n/i18n';
import { computed, ref } from 'vue';
import type { CompareCaseRow } from '../../composables/useCompareCases';
import type { CompareMetricGroup, CompareVersion } from '../../composables/useCompareData';
import { deriveRunsStatus } from '../../evaluation.utils';
import CasesTable from './CasesTable.vue';
import OutputsTab from './OutputsTab.vue';
import MetricsTab from './MetricsTab.vue';
import WorkflowDiffTab from './WorkflowDiffTab.vue';
const props = defineProps<{
versions: CompareVersion[];
metricGroups: CompareMetricGroup[];
caseRows: CompareCaseRow[];
casesLoading: boolean;
casesError?: boolean;
workflowId: string;
// metric name → its custom LLM-judge prompt, when configured.
metricPrompts?: Record<string, string>;
}>();
const i18n = useI18n();
type TabValue = 'cases' | 'outputs' | 'metrics' | 'workflowDiff';
const activeTab = ref<TabValue>('cases');
const selectedCaseIndex = ref(0);
const tabs = computed<Array<TabOptions<TabValue>>>(() => [
{ value: 'cases', label: i18n.baseText('evaluation.compare.tabs.cases') },
{ value: 'outputs', label: i18n.baseText('evaluation.compare.tabs.outputs') },
{ value: 'metrics', label: i18n.baseText('evaluation.compare.tabs.metrics') },
{ value: 'workflowDiff', label: i18n.baseText('evaluation.compare.tabs.workflowDiff') },
]);
const hasCases = computed(() => props.caseRows.length > 0);
// While any version is still executing, per-case scores stream in — surface an
// in-progress note so the partially-filled table doesn't read as broken.
const isRunning = computed(() => deriveRunsStatus(props.versions) === 'running');
// Drilling into a case row jumps to its side-by-side outputs — the same detail
// a per-case drawer would show, without a second surface to keep in sync.
function onDrilldown(caseIndex: number) {
selectedCaseIndex.value = caseIndex;
activeTab.value = 'outputs';
}
</script>
<template>
<section :class="$style.tabs" data-test-id="compare-tabs">
<N8nTabs v-model="activeTab" :options="tabs" />
<N8nText v-if="casesError" size="xsmall" color="danger" data-test-id="compare-cases-error">
{{ i18n.baseText('evaluation.compare.cases.loadError') }}
</N8nText>
<div :class="$style.panel">
<template v-if="activeTab === 'cases'">
<!-- Gate on loading first: the per-version case fetches resolve
independently, so rendering mid-load would flash partial rows
and a false dataset mismatch until every run settles. -->
<N8nText v-if="casesLoading" size="small" color="text-light">
{{ i18n.baseText('evaluation.compare.cases.loading') }}
</N8nText>
<N8nText v-else-if="!hasCases" size="small" color="text-light">
{{ i18n.baseText('evaluation.compare.cases.empty') }}
</N8nText>
<template v-else>
<N8nText
v-if="isRunning"
size="xsmall"
color="text-light"
:class="$style.runningNote"
data-test-id="compare-cases-running"
>
{{ i18n.baseText('evaluation.compare.cases.running') }}
</N8nText>
<CasesTable :versions="versions" :case-rows="caseRows" @drilldown="onDrilldown" />
</template>
</template>
<template v-else-if="activeTab === 'outputs'">
<N8nText v-if="casesLoading" size="small" color="text-light">
{{ i18n.baseText('evaluation.compare.cases.loading') }}
</N8nText>
<OutputsTab
v-else
:versions="versions"
:case-rows="caseRows"
:selected-index="selectedCaseIndex"
@update:selected-index="selectedCaseIndex = $event"
/>
</template>
<MetricsTab
v-else-if="activeTab === 'metrics'"
:versions="versions"
:metric-groups="metricGroups"
:metric-prompts="metricPrompts"
/>
<WorkflowDiffTab v-else :versions="versions" :workflow-id="workflowId" />
</div>
</section>
</template>
<style module lang="scss">
.tabs {
display: flex;
flex-direction: column;
gap: var(--spacing--sm);
}
.panel {
min-height: 120px;
}
.runningNote {
display: block;
margin-bottom: var(--spacing--2xs);
}
</style>
@@ -0,0 +1,19 @@
import { describe, expect, it } from 'vitest';
import { createComponentRenderer } from '@/__tests__/render';
import DatasetMismatchBanner from './DatasetMismatchBanner.vue';
const renderComponent = createComponentRenderer(DatasetMismatchBanner);
describe('DatasetMismatchBanner', () => {
it('renders the per-version case counts in the warning', () => {
const { container } = renderComponent({
props: { mismatch: { hasMismatch: true, counts: [12, 12, 10], maxCount: 12 } },
});
const banner = container.querySelector('[data-test-id="compare-dataset-mismatch"]');
expect(banner).not.toBeNull();
expect(banner?.textContent).toContain('12, 12, 10');
});
});
@@ -0,0 +1,25 @@
<script setup lang="ts">
import { N8nCallout } from '@n8n/design-system';
import { useI18n } from '@n8n/i18n';
import { computed } from 'vue';
import type { DatasetMismatch } from '../../composables/useCompareCases';
const props = defineProps<{
mismatch: DatasetMismatch;
}>();
const i18n = useI18n();
const countsLabel = computed(() => props.mismatch.counts.join(', '));
</script>
<template>
<N8nCallout theme="warning" icon="triangle-alert" data-test-id="compare-dataset-mismatch">
{{
i18n.baseText('evaluation.compare.datasetMismatch', {
interpolate: { counts: countsLabel },
})
}}
</N8nCallout>
</template>
@@ -0,0 +1,97 @@
<script setup lang="ts">
import { N8nText } from '@n8n/design-system';
import { useI18n } from '@n8n/i18n';
import { computed, ref } from 'vue';
import { getMetricDescriptionKey } from '../../evaluation.utils';
const props = defineProps<{
metricKey: string;
// The metric's custom LLM-judge prompt — the specific criteria the user
// configured (e.g. "checks for markdown"). Omitted for preset/non-judge
// metrics, which only show the generic description.
prompt?: string;
}>();
const i18n = useI18n();
const expanded = ref(false);
// Characters of the custom criteria shown before it's truncated behind "Show
// more" — judging prompts are often long rubrics.
const PREVIEW_CHARS = 120;
const description = computed(() => {
const key = getMetricDescriptionKey(props.metricKey);
return key ? i18n.baseText(key) : '';
});
const isLong = computed(() => (props.prompt?.length ?? 0) > PREVIEW_CHARS);
const promptText = computed(() => {
if (!props.prompt) return '';
if (expanded.value || !isLong.value) return props.prompt;
return `${props.prompt.slice(0, PREVIEW_CHARS).trimEnd()}…`;
});
</script>
<template>
<div v-if="description || prompt" :class="$style.wrap" data-test-id="metric-criteria">
<N8nText v-if="description" size="xsmall" color="text-light" :class="$style.text">
{{ description }}
</N8nText>
<N8nText
v-if="prompt"
size="xsmall"
color="text-light"
:class="[$style.text, expanded ? $style.expanded : null]"
>
<span :class="$style.label">{{ i18n.baseText('evaluation.metric.criteria.label') }}</span>
{{ promptText }}
<button
v-if="isLong"
type="button"
:class="$style.toggle"
data-test-id="metric-criteria-toggle"
@click.stop="expanded = !expanded"
>
{{
expanded
? i18n.baseText('evaluation.metric.criteria.showLess')
: i18n.baseText('evaluation.metric.criteria.showMore')
}}
</button>
</N8nText>
</div>
</template>
<style module lang="scss">
.wrap {
display: flex;
flex-direction: column;
gap: var(--spacing--5xs);
max-width: 320px;
}
.text {
line-height: 1.3;
}
.expanded {
white-space: pre-wrap;
word-break: break-word;
}
.label {
font-weight: var(--font-weight--bold);
}
.toggle {
border: none;
background: none;
padding: 0;
margin-left: var(--spacing--4xs);
color: var(--color--primary);
cursor: pointer;
font: inherit;
}
</style>
@@ -0,0 +1,102 @@
<script setup lang="ts">
import { N8nText } from '@n8n/design-system';
import { useI18n } from '@n8n/i18n';
import type { CompareMetricGroup, CompareVersion } from '../../composables/useCompareData';
import { formatMetricPercent } from '../../evaluation.utils';
import { versionColorVar } from '../shared/versionPalette';
import MetricCriteria from './MetricCriteria.vue';
defineProps<{
versions: CompareVersion[];
metricGroups: CompareMetricGroup[];
// metric name → its custom LLM-judge prompt, when configured.
metricPrompts?: Record<string, string>;
}>();
const i18n = useI18n();
</script>
<template>
<div data-test-id="compare-metrics-tab">
<table v-if="metricGroups.length > 0" :class="$style.table">
<thead>
<tr>
<th>{{ i18n.baseText('evaluation.compare.metrics.col.metric') }}</th>
<th v-for="version in versions" :key="version.testRunId" :class="$style.value">
<span :class="$style.head">
<span :class="$style.dot" :style="{ background: versionColorVar(version.index) }" />
{{ version.letter }}
</span>
</th>
</tr>
</thead>
<tbody>
<tr v-for="group in metricGroups" :key="group.key" data-test-id="compare-metrics-row">
<td>
<div :class="$style.metric">
<N8nText size="xsmall">{{ group.label }}</N8nText>
<MetricCriteria :metric-key="group.key" :prompt="metricPrompts?.[group.key]" />
</div>
</td>
<td
v-for="(value, versionIndex) in group.values"
:key="versionIndex"
:class="$style.value"
>
<N8nText size="xsmall" :bold="versionIndex === group.bestIndex">
{{ formatMetricPercent(value ?? undefined) }}
</N8nText>
</td>
</tr>
</tbody>
</table>
<N8nText v-else size="small" color="text-light">
{{ i18n.baseText('evaluation.compare.metrics.empty') }}
</N8nText>
</div>
</template>
<style module lang="scss">
.table {
width: 100%;
border-collapse: collapse;
font-size: var(--font-size--xs);
th,
td {
text-align: left;
padding: var(--spacing--2xs) var(--spacing--xs);
border-bottom: 1px solid var(--border-color--subtle);
}
th {
color: var(--text-color--subtle);
font-weight: var(--font-weight--medium);
}
}
.value {
text-align: center;
vertical-align: top;
}
.metric {
display: flex;
flex-direction: column;
gap: var(--spacing--5xs);
max-width: 320px;
}
.head {
display: inline-flex;
align-items: center;
gap: var(--spacing--4xs);
}
.dot {
width: 8px;
height: 8px;
border-radius: var(--radius--full);
}
</style>
@@ -0,0 +1,79 @@
import { fireEvent } from '@testing-library/vue';
import { describe, expect, it } from 'vitest';
import { createComponentRenderer } from '@/__tests__/render';
import type { CompareCaseCell, CompareCaseRow } from '../../composables/useCompareCases';
import type { CompareVersion } from '../../composables/useCompareData';
import OutputsTab from './OutputsTab.vue';
const versions: CompareVersion[] = [
{
index: 0,
testRunId: 'run-a',
workflowVersionId: 'v0',
letter: 'A',
label: 'Baseline',
status: 'completed',
avgScore: null,
},
{
index: 1,
testRunId: 'run-b',
workflowVersionId: 'v1',
letter: 'B',
label: 'Candidate',
status: 'completed',
avgScore: null,
},
];
const cell = (versionIndex: number, output: string): CompareCaseCell => ({
versionIndex,
testCaseId: `c${versionIndex}`,
inputs: { q: 'What is 2+2?' },
outputs: { output },
metrics: { helpfulness: 0.8 },
score: 0.8,
});
const rows: CompareCaseRow[] = [
{
index: 0,
displayIndex: 1,
inputPreview: 'What is 2+2?',
cells: [cell(0, 'four'), cell(1, 'the answer is four')],
bestVersionIndex: 1,
},
{
index: 1,
displayIndex: 2,
inputPreview: 'Capital of France?',
cells: [cell(0, 'Paris'), cell(1, 'Paris, France')],
bestVersionIndex: 1,
},
];
const renderComponent = createComponentRenderer(OutputsTab);
describe('OutputsTab', () => {
it('renders one output column per version for the selected case', () => {
const { container } = renderComponent({
props: { versions, caseRows: rows, selectedIndex: 0 },
});
expect(container.querySelectorAll('[data-test-id="compare-outputs-column"]')).toHaveLength(2);
expect(container.textContent).toContain('four');
expect(container.textContent).toContain('the answer is four');
});
it('emits the new index when a sidebar case is clicked', async () => {
const { container, emitted } = renderComponent({
props: { versions, caseRows: rows, selectedIndex: 0 },
});
const items = container.querySelectorAll('[data-test-id="compare-outputs-case"]');
await fireEvent.click(items[1]);
expect(emitted()['update:selectedIndex']).toEqual([[1]]);
});
});
@@ -0,0 +1,220 @@
<script setup lang="ts">
import { N8nText } from '@n8n/design-system';
import { useI18n } from '@n8n/i18n';
import { computed } from 'vue';
import type { CompareCaseRow } from '../../composables/useCompareCases';
import type { CompareVersion } from '../../composables/useCompareData';
import {
extractAnswerText,
formatMetricLabel,
formatMetricPercent,
getMetricCategory,
getUserDefinedMetricNames,
} from '../../evaluation.utils';
import VersionAvatar from '../shared/VersionAvatar.vue';
const props = defineProps<{
versions: CompareVersion[];
caseRows: CompareCaseRow[];
selectedIndex: number;
}>();
const emit = defineEmits<{
'update:selectedIndex': [index: number];
}>();
const i18n = useI18n();
const selectedRow = computed(
() => props.caseRows.find((row) => row.index === props.selectedIndex) ?? props.caseRows[0],
);
function metricEntries(metrics: Record<string, number> | undefined) {
if (!metrics) return [];
return getUserDefinedMetricNames(metrics).map((key) => ({
key,
label: formatMetricLabel(key),
value: formatMetricPercent(metrics[key], { category: getMetricCategory(key) }),
}));
}
</script>
<template>
<div :class="$style.outputs" data-test-id="compare-outputs-tab">
<aside :class="$style.sidebar">
<N8nText size="xsmall" bold color="text-light" :class="$style.sidebarTitle">
{{ i18n.baseText('evaluation.compare.outputs.casesSidebarTitle') }}
</N8nText>
<button
v-for="row in caseRows"
:key="row.index"
type="button"
:class="[$style.caseItem, { [$style.caseItemActive]: row.index === selectedRow?.index }]"
data-test-id="compare-outputs-case"
@click="emit('update:selectedIndex', row.index)"
>
<N8nText size="xsmall" bold>#{{ row.displayIndex }}</N8nText>
<N8nText size="xsmall" color="text-light" :class="$style.caseItemInput">
{{ row.inputPreview }}
</N8nText>
</button>
</aside>
<section v-if="selectedRow" :class="$style.main">
<div :class="$style.inputRow">
<N8nText size="xsmall" bold color="text-light">
{{ i18n.baseText('evaluation.compare.outputs.input') }}
</N8nText>
<N8nText size="small">{{ selectedRow.inputPreview }}</N8nText>
</div>
<div :class="$style.columns">
<article
v-for="cell in selectedRow.cells"
:key="cell.versionIndex"
:class="$style.column"
data-test-id="compare-outputs-column"
>
<header :class="$style.columnHeader">
<VersionAvatar :index="cell.versionIndex" variant="square" size="small" />
<N8nText size="xsmall" color="text-light">
{{ versions[cell.versionIndex]?.label }}
</N8nText>
</header>
<div :class="$style.answer">
<N8nText v-if="cell.outputs !== undefined" size="small">
{{ extractAnswerText(cell.outputs) }}
</N8nText>
<N8nText v-else size="small" color="text-light">
{{ i18n.baseText('evaluation.compare.outputs.noOutput') }}
</N8nText>
</div>
<footer :class="$style.metrics">
<span
v-for="metric in metricEntries(cell.metrics)"
:key="metric.key"
:class="$style.metric"
>
<N8nText size="xsmall" color="text-light">{{ metric.label }}</N8nText>
<N8nText size="xsmall" bold>{{ metric.value }}</N8nText>
</span>
</footer>
</article>
</div>
</section>
</div>
</template>
<style module lang="scss">
.outputs {
display: flex;
gap: var(--spacing--md);
align-items: flex-start;
}
.sidebar {
width: 240px;
flex-shrink: 0;
display: flex;
flex-direction: column;
gap: var(--spacing--4xs);
max-height: 480px;
overflow-y: auto;
}
.sidebarTitle {
padding: var(--spacing--2xs) var(--spacing--xs);
}
.caseItem {
display: flex;
flex-direction: column;
gap: var(--spacing--5xs);
align-items: flex-start;
text-align: left;
padding: var(--spacing--2xs) var(--spacing--xs);
border: none;
border-radius: var(--radius);
background: transparent;
cursor: pointer;
&:hover {
background: var(--background--subtle);
}
}
.caseItemActive {
background: var(--background--subtle);
}
.caseItemInput {
max-width: 100%;
overflow: hidden;
text-overflow: ellipsis;
white-space: nowrap;
}
.main {
flex: 1;
min-width: 0;
display: flex;
flex-direction: column;
gap: var(--spacing--sm);
}
.inputRow {
display: flex;
flex-direction: column;
gap: var(--spacing--4xs);
padding: var(--spacing--sm);
border-radius: var(--radius);
background: var(--background--subtle);
}
.columns {
display: flex;
gap: var(--spacing--sm);
overflow-x: auto;
}
.column {
flex: 1;
min-width: 220px;
display: flex;
flex-direction: column;
gap: var(--spacing--2xs);
padding: var(--spacing--sm);
border: 1px solid var(--border-color--subtle);
border-radius: var(--radius--md);
}
.columnHeader {
display: inline-flex;
align-items: center;
gap: var(--spacing--2xs);
}
.answer {
max-height: 240px;
overflow-y: auto;
white-space: pre-wrap;
word-break: break-word;
}
.metrics {
display: flex;
flex-wrap: wrap;
gap: var(--spacing--2xs);
padding-top: var(--spacing--2xs);
border-top: 1px solid var(--border-color--subtle);
}
.metric {
display: inline-flex;
gap: var(--spacing--4xs);
align-items: baseline;
}
</style>
@@ -5,10 +5,13 @@ import { computed, ref } from 'vue';
import type { CompareMetricGroup, CompareVersion } from '../../composables/useCompareData';
import GroupedMetricChart from '../shared/GroupedMetricChart.vue';
import MetricCriteria from './MetricCriteria.vue';
const props = defineProps<{
metricGroups: CompareMetricGroup[];
versions: CompareVersion[];
// metric name → its custom LLM-judge prompt, when configured.
metricPrompts?: Record<string, string>;
}>();
const i18n = useI18n();
@@ -59,6 +62,7 @@ const letters = computed(() => props.versions.map((version) => version.letter));
<N8nText size="small" bold color="text-base" :class="$style.panelHeading">
{{ group.label }}
</N8nText>
<MetricCriteria :metric-key="group.key" :prompt="metricPrompts?.[group.key]" />
<GroupedMetricChart
variant="detailed"
:groups="[{ label: group.label, values: group.values, letters }]"
@@ -0,0 +1,74 @@
import { waitFor } from '@testing-library/vue';
import { describe, expect, it, vi } from 'vitest';
import { createComponentRenderer } from '@/__tests__/render';
import type { CompareVersion } from '../../composables/useCompareData';
import WorkflowDiffTab from './WorkflowDiffTab.vue';
const fetchWorkflow = vi.fn();
const getWorkflowVersion = vi.fn();
vi.mock('@/app/stores/workflowsList.store', () => ({
useWorkflowsListStore: () => ({ fetchWorkflow }),
}));
vi.mock('@/features/workflows/workflowHistory/workflowHistory.store', () => ({
useWorkflowHistoryStore: () => ({ getWorkflowVersion }),
}));
vi.mock('@/app/composables/useToast', () => ({
useToast: () => ({ showError: vi.fn() }),
}));
// Stub the heavy diff canvas — this tab only wires data into it.
vi.mock('@/features/workflows/workflowDiff/WorkflowDiffView.vue', () => ({
default: { name: 'WorkflowDiffView', template: '<div data-test-id="wf-diff-stub" />' },
}));
const version = (over: Partial<CompareVersion>): CompareVersion => ({
index: 0,
testRunId: 'run',
workflowVersionId: 'v0',
letter: 'A',
label: 'Baseline',
status: 'completed',
avgScore: null,
...over,
});
const renderComponent = createComponentRenderer(WorkflowDiffTab);
describe('WorkflowDiffTab', () => {
beforeEach(() => {
fetchWorkflow.mockReset().mockResolvedValue({ id: 'wf-1', nodes: [], connections: {} });
getWorkflowVersion.mockReset().mockResolvedValue({
versionId: 'v1',
workflowId: 'wf-1',
nodes: [],
connections: {},
nodeGroups: [],
});
});
it('prompts for a second version and skips loading when only one is present', () => {
const { queryByTestId } = renderComponent({
props: { versions: [version({ index: 0 })], workflowId: 'wf-1' },
});
expect(queryByTestId('workflow-diff-source-select')).toBeNull();
expect(fetchWorkflow).not.toHaveBeenCalled();
});
it('resolves the current draft from the base workflow and a version from its snapshot', async () => {
const versions = [
version({ index: 0, workflowVersionId: null, letter: 'A', label: 'Current draft' }),
version({ index: 1, workflowVersionId: 'v1', letter: 'B', label: 'v1' }),
];
const { findByTestId } = renderComponent({ props: { versions, workflowId: 'wf-1' } });
await findByTestId('wf-diff-stub');
await waitFor(() => expect(fetchWorkflow).toHaveBeenCalledWith('wf-1'));
// Only the real version pulls a history snapshot; the draft reuses the base.
expect(getWorkflowVersion).toHaveBeenCalledTimes(1);
expect(getWorkflowVersion).toHaveBeenCalledWith('wf-1', 'v1');
});
});
@@ -0,0 +1,227 @@
<script setup lang="ts">
import { N8nOption, N8nSelect, N8nText } from '@n8n/design-system';
import { useI18n } from '@n8n/i18n';
import omit from 'lodash/omit';
import { deepCopy } from 'n8n-workflow';
import { computed, markRaw, ref, watch } from 'vue';
import { useToast } from '@/app/composables/useToast';
import { useWorkflowsListStore } from '@/app/stores/workflowsList.store';
import WorkflowDiffView from '@/features/workflows/workflowDiff/WorkflowDiffView.vue';
import { useWorkflowHistoryStore } from '@/features/workflows/workflowHistory/workflowHistory.store';
import type { IWorkflowDb } from '@/Interface';
import type { CompareVersion } from '../../composables/useCompareData';
const props = defineProps<{
versions: CompareVersion[];
workflowId: string;
}>();
const i18n = useI18n();
const toast = useToast();
const workflowsListStore = useWorkflowsListStore();
const workflowHistoryStore = useWorkflowHistoryStore();
// The compare view can hold >2 versions but a diff is pairwise, so the tab
// picks two sides (defaulting to the first two) and lets the user re-point
// either. Versions are keyed by their stable compare `index`.
const sourceIndex = ref(props.versions[0]?.index ?? 0);
const targetIndex = ref(props.versions[1]?.index ?? props.versions[0]?.index ?? 0);
const sourceWorkflow = ref<IWorkflowDb>();
const targetWorkflow = ref<IWorkflowDb>();
const isLoading = ref(false);
const canDiff = computed(() => props.versions.length >= 2);
const versionOptions = computed(() =>
props.versions.map((version) => ({
value: version.index,
label: `${version.letter} · ${version.label}`,
})),
);
const versionByIndex = (index: number): CompareVersion | undefined =>
props.versions.find((version) => version.index === index);
const labelFor = (index: number): string => {
const version = versionByIndex(index);
return version ? `${version.letter} · ${version.label}` : '';
};
// Stale-response guard: only the newest load applies its result.
let loadRequestId = 0;
// Hand the diff a fully-detached, non-reactive copy. WorkflowDiffView tracks the
// workflow deeply and its render stores snap node positions in place; on a
// reactive source that write feeds back into a hydrate → dispose → hydrate loop
// that pins the main thread. A plain, `markRaw`'d clone breaks the cycle — the
// diff is read-only, so it never needs reactivity or to reach the live stores.
const detach = (workflow: IWorkflowDb): IWorkflowDb => markRaw(deepCopy(workflow));
// Resolve a version to a full workflow the diff can render. A null
// `workflowVersionId` is the "current draft" — the live workflow itself; a real
// id pulls that history snapshot's nodes/connections onto the base workflow so
// settings and metadata still line up with the current workflow.
const resolveWorkflow = async (
base: IWorkflowDb,
version: CompareVersion,
): Promise<IWorkflowDb> => {
const bare = omit(base, 'pinData');
if (version.workflowVersionId === null) return bare;
const snapshot = await workflowHistoryStore.getWorkflowVersion(
props.workflowId,
version.workflowVersionId,
);
return {
...bare,
versionId: snapshot.versionId,
nodes: snapshot.nodes,
connections: snapshot.connections,
nodeGroups: snapshot.nodeGroups ?? [],
};
};
const load = async () => {
// Nothing to diff until there are two versions; the template shows a prompt
// instead, so don't fetch workflows we won't render.
if (!canDiff.value) return;
const source = versionByIndex(sourceIndex.value);
const target = versionByIndex(targetIndex.value);
if (!source || !target) return;
const requestId = ++loadRequestId;
isLoading.value = true;
try {
const base = await workflowsListStore.fetchWorkflow(props.workflowId);
const [resolvedSource, resolvedTarget] = await Promise.all([
resolveWorkflow(base, source),
resolveWorkflow(base, target),
]);
if (requestId !== loadRequestId) return;
sourceWorkflow.value = detach(resolvedSource);
targetWorkflow.value = detach(resolvedTarget);
} catch (error) {
toast.showError(error, i18n.baseText('evaluation.compare.workflowDiff.loadError'));
} finally {
if (requestId === loadRequestId) isLoading.value = false;
}
};
// Keep the two sides distinct: picking a version already on the other side
// swaps them rather than diffing a version against itself.
const onSourceChange = (next: number) => {
if (next === targetIndex.value) targetIndex.value = sourceIndex.value;
sourceIndex.value = next;
};
const onTargetChange = (next: number) => {
if (next === sourceIndex.value) sourceIndex.value = targetIndex.value;
targetIndex.value = next;
};
watch([sourceIndex, targetIndex], load, { immediate: true });
</script>
<template>
<div data-test-id="compare-workflow-diff-tab">
<div v-if="!canDiff" :class="$style.placeholder">
<N8nText size="small" color="text-light">
{{ i18n.baseText('evaluation.compare.workflowDiff.needTwo') }}
</N8nText>
</div>
<template v-else>
<div :class="$style.controls">
<label :class="$style.control">
<N8nText size="xsmall" color="text-light" bold>
{{ i18n.baseText('evaluation.compare.workflowDiff.base') }}
</N8nText>
<N8nSelect
:model-value="sourceIndex"
size="small"
data-test-id="workflow-diff-source-select"
@update:model-value="onSourceChange(Number($event))"
>
<N8nOption
v-for="opt in versionOptions"
:key="opt.value"
:value="opt.value"
:label="opt.label"
/>
</N8nSelect>
</label>
<label :class="$style.control">
<N8nText size="xsmall" color="text-light" bold>
{{ i18n.baseText('evaluation.compare.workflowDiff.compare') }}
</N8nText>
<N8nSelect
:model-value="targetIndex"
size="small"
data-test-id="workflow-diff-target-select"
@update:model-value="onTargetChange(Number($event))"
>
<N8nOption
v-for="opt in versionOptions"
:key="opt.value"
:value="opt.value"
:label="opt.label"
/>
</N8nSelect>
</label>
</div>
<div :class="$style.diff">
<div v-if="isLoading" :class="$style.state">
<N8nText size="small" color="text-light">{{ i18n.baseText('generic.loading') }}</N8nText>
</div>
<WorkflowDiffView
v-else-if="sourceWorkflow && targetWorkflow"
:source-workflow="sourceWorkflow"
:target-workflow="targetWorkflow"
:source-label="labelFor(sourceIndex)"
:target-label="labelFor(targetIndex)"
/>
</div>
</template>
</div>
</template>
<style module lang="scss">
.placeholder {
display: flex;
align-items: center;
justify-content: center;
padding: var(--spacing--2xl);
border: 1px dashed var(--border-color--subtle);
border-radius: var(--radius--md);
}
.controls {
display: flex;
gap: var(--spacing--md);
margin-bottom: var(--spacing--sm);
}
.control {
display: flex;
flex-direction: column;
gap: var(--spacing--4xs);
min-width: 220px;
}
.diff {
height: 70vh;
min-height: 480px;
border: 1px solid var(--border-color--subtle);
border-radius: var(--radius--md);
overflow: hidden;
}
.state {
display: flex;
align-items: center;
justify-content: center;
height: 100%;
}
</style>
@@ -36,10 +36,33 @@ const openCompare = () => {
// `null` until the detail (with run statuses) has loaded — the list view only
// pre-fetches detail for the first few cards and lazy-loads the rest on hover,
// so we must not assert "Done" for a card whose runs might still be in flight.
const status = computed<'done' | 'running' | null>(() =>
const status = computed<'done' | 'running' | 'error' | null>(() =>
props.detail ? deriveRunsStatus(props.detail.runs) : null,
);
// Badge theme + label per status; `null` while detail is still loading (no badge).
const statusBadge = computed(() => {
switch (status.value) {
case 'done':
return {
theme: 'success' as const,
label: i18n.baseText('evaluation.collections.card.done'),
};
case 'running':
return {
theme: 'tertiary' as const,
label: i18n.baseText('evaluation.collections.card.running'),
};
case 'error':
return {
theme: 'warning' as const,
label: i18n.baseText('evaluation.collections.card.failed'),
};
default:
return null;
}
});
// Append a right arrow so the CTA reads "Open compare →" the way the
// Figma mock does. N8nButton doesn't accept a trailing icon prop today,
// so the arrow lives in the label string.
@@ -145,14 +168,8 @@ onMounted(() => observe(cardRef.value));
<div :class="$style.cardHeader">
<div :class="$style.cardHeading">
<N8nText size="medium" bold>{{ collection.name }}</N8nText>
<N8nBadge v-if="status" :theme="status === 'done' ? 'success' : 'tertiary'" size="small">
{{
i18n.baseText(
status === 'done'
? 'evaluation.collections.card.done'
: 'evaluation.collections.card.running',
)
}}
<N8nBadge v-if="statusBadge" :theme="statusBadge.theme" size="small">
{{ statusBadge.label }}
</N8nBadge>
</div>
<N8nText size="xsmall" color="text-light">
@@ -4,7 +4,7 @@ import { useI18n, type BaseTextKey } from '@n8n/i18n';
import { computed } from 'vue';
import type { TestRunRecord } from '../../evaluation.api';
import { isScoreShapedMetric } from '../../evaluation.utils';
import { averageNormalizedScore } from '../../evaluation.utils';
const STATUS_PILL_THEME: Record<string, 'success' | 'warning' | 'danger' | 'tertiary'> = {
completed: 'success',
@@ -35,16 +35,13 @@ const props = defineProps<{
const i18n = useI18n();
// Average only score-shaped metrics (values in [0, 1]). Eval-config metrics
// commonly co-exist with absolute counts (tokens, latency_ms) in the same
// `metrics` map, and naively averaging across all of them produces nonsense
// like `198431%` (mostly the token total).
// Average the run's score metrics, each normalized to [0, 1] by its scale.
// Eval-config metrics commonly co-exist with absolute counts (tokens,
// latency_ms) in the same `metrics` map; those aren't scores and are dropped,
// so a naive all-metric average can't produce nonsense like `198431%`.
const score = computed<number | null>(() => {
const m = props.run.metrics;
if (!m) return null;
const values = Object.values(m).filter(isScoreShapedMetric);
if (values.length === 0) return null;
return Math.round((values.reduce((a, b) => a + b, 0) / values.length) * 100);
const avg = averageNormalizedScore(props.run.metrics);
return avg === null ? null : Math.round(avg * 100);
});
const statusTheme = computed(() => STATUS_PILL_THEME[props.run.status] ?? 'tertiary');
@@ -2,15 +2,19 @@
import type { TestRunRecord } from '../../evaluation.api';
import MetricsChart from './MetricsChart.vue';
import TestRunsTable from './TestRunsTable.vue';
import { N8nPagination } from '@n8n/design-system';
import { useI18n } from '@n8n/i18n';
import { VIEWS } from '@/app/constants';
import { convertToDisplayDate } from '@/app/utils/formatters/dateFormatter';
import { computed } from 'vue';
import { computed, ref, watch } from 'vue';
import { useRouter } from 'vue-router';
const props = defineProps<{
runs: Array<TestRunRecord & { index: number }>;
workflowId: string;
// When set, the table paginates to this many rows per page. The chart still
// plots every run. Omitted → the full table renders, no pagination.
pageSize?: number;
}>();
const locale = useI18n();
@@ -18,6 +22,35 @@ const router = useRouter();
const selectedMetric = defineModel<string>('selectedMetric', { required: true });
const currentPage = ref(1);
const showPagination = computed(() => !!props.pageSize && props.runs.length > props.pageSize);
// Newest-first ordering for pagination so page 1 holds the most recent runs
// (the incoming `runs` prop is ascending, for the chart's left→right trend).
// Ties on `runAt` fall back to run number so equal-timestamp runs stay in
// sequence. The table re-applies its own descending sort within each page.
const runsNewestFirst = computed(() =>
[...props.runs].sort((a, b) => {
const byDate = new Date(b.runAt).getTime() - new Date(a.runAt).getTime();
return byDate !== 0 ? byDate : b.index - a.index;
}),
);
const pagedRuns = computed(() => {
if (!props.pageSize) return props.runs;
const start = (currentPage.value - 1) * props.pageSize;
return runsNewestFirst.value.slice(start, start + props.pageSize);
});
// Clamp the page when the run set shrinks (or polling changes it) so we never
// land on an empty page past the end.
watch(
() => props.runs.length,
() => {
const maxPage = props.pageSize ? Math.max(1, Math.ceil(props.runs.length / props.pageSize)) : 1;
if (currentPage.value > maxPage) currentPage.value = maxPage;
},
);
const metrics = computed(() => {
const metricKeys = props.runs.reduce((acc, run) => {
Object.keys(run.metrics ?? {}).forEach((metric) => acc.add(metric));
@@ -74,17 +107,27 @@ const handleRowClick = (row: TestRunRecord) => {
</script>
<template>
<div :class="$style.runs">
<div :class="[$style.runs, pageSize ? $style.paged : null]">
<MetricsChart v-model:selected-metric="selectedMetric" :runs="runs" />
<TestRunsTable
:class="$style.runsTable"
:runs
:runs="pagedRuns"
:columns
:selectable="true"
data-test-id="past-runs-table"
@row-click="handleRowClick"
/>
<N8nPagination
v-if="showPagination"
:class="$style.pagination"
layout="prev, pager, next"
:current-page="currentPage"
:page-size="pageSize"
:total="runs.length"
@update:current-page="currentPage = $event"
/>
</div>
</template>
@@ -97,4 +140,18 @@ const handleRowClick = (row: TestRunRecord) => {
overflow: auto;
margin-bottom: 20px;
}
// Stacked/paginated mode: the table is capped to `pageSize` rows so it never
// needs its own scroll region. Let the section grow with its content and defer
// scrolling to the page so there's a single page-level scrollbar.
.paged {
flex: none;
overflow: visible;
margin-bottom: 0;
}
.pagination {
display: flex;
justify-content: center;
}
</style>
@@ -48,6 +48,6 @@ describe('GroupedMetricChart', () => {
// 0.9 ≥ 0.6 keeps its version color; 0.4 < 0.6 flips to danger.
expect(bars[0].getAttribute('fill')).toBe(versionColorVar(0));
expect(bars[1].getAttribute('fill')).toBe('var(--color--red-700)');
expect(bars[1].getAttribute('fill')).toBe('var(--icon-color--danger)');
});
});
@@ -47,7 +47,9 @@ const GEOMETRY = {
const geo = computed(() => GEOMETRY[props.variant]);
const CRITICAL_COLOR = 'var(--color--red-700)';
// Semantic danger token (not the `--color--red-*` primitive scale): it adapts
// to dark theme and stays distinct from the version palette's red hue.
const CRITICAL_COLOR = 'var(--icon-color--danger)';
const clamp01 = (v: number, max: number) => {
if (!Number.isFinite(v)) return 0;
@@ -0,0 +1,193 @@
import { createTestingPinia } from '@pinia/testing';
import { setActivePinia } from 'pinia';
import { beforeEach, describe, expect, it, vi } from 'vitest';
import { nextTick, ref } from 'vue';
import type { TestCaseExecutionRecord } from '../evaluation.api';
import { useEvaluationStore } from '../evaluation.store';
import type { EvaluationCollectionDetail } from '../evalCollections.types';
import { useCompareCases } from './useCompareCases';
const run = (testRunId: string): EvaluationCollectionDetail['runs'][number] => ({
testRunId,
workflowVersionId: testRunId,
status: 'completed',
runAt: null,
completedAt: null,
avgScore: null,
metrics: null,
});
const detailWith = (runIds: string[]): EvaluationCollectionDetail => ({
id: 'col-1',
name: 'Compare',
description: null,
workflowId: 'wf-1',
evaluationConfigId: 'cfg-1',
createdById: 'u1',
createdAt: '',
updatedAt: '',
runCount: runIds.length,
runs: runIds.map(run),
});
const caseRecord = (
overrides: Partial<TestCaseExecutionRecord> & Pick<TestCaseExecutionRecord, 'id' | 'testRunId'>,
): TestCaseExecutionRecord => ({
executionId: null,
status: 'success',
createdAt: '',
updatedAt: '',
runAt: null,
...overrides,
});
describe('useCompareCases', () => {
let store: ReturnType<typeof useEvaluationStore>;
beforeEach(() => {
setActivePinia(createTestingPinia({ stubActions: false, createSpy: vi.fn }));
store = useEvaluationStore();
store.fetchTestCaseExecutions = vi.fn(
async () => [],
) as unknown as typeof store.fetchTestCaseExecutions;
});
function seed(records: TestCaseExecutionRecord[]) {
store.$patch((state) => {
state.testCaseExecutionsById = Object.fromEntries(records.map((r) => [r.id, r]));
});
}
it('aligns cases across runs by runIndex and picks the best version per case', async () => {
// helpfulness is a 1–5 AI-judge metric → normalized /5 (3.5 → 0.7 etc.).
seed([
caseRecord({ id: 'a0', testRunId: 'run-a', runIndex: 0, metrics: { helpfulness: 3.5 } }),
caseRecord({ id: 'a1', testRunId: 'run-a', runIndex: 1, metrics: { helpfulness: 2.5 } }),
// run-b returned out of order — alignment must sort by runIndex.
caseRecord({ id: 'b1', testRunId: 'run-b', runIndex: 1, metrics: { helpfulness: 4.5 } }),
caseRecord({ id: 'b0', testRunId: 'run-b', runIndex: 0, metrics: { helpfulness: 3 } }),
]);
const { caseRows, mismatch } = useCompareCases(
ref(detailWith(['run-a', 'run-b'])),
ref('wf-1'),
);
await nextTick();
expect(mismatch.value.hasMismatch).toBe(false);
expect(caseRows.value).toHaveLength(2);
// case #0: A 0.7 vs B 0.6 → A best
expect(caseRows.value[0].bestVersionIndex).toBe(0);
expect(caseRows.value[0].cells[1].testCaseId).toBe('b0');
// case #1: A 0.5 vs B 0.9 → B best
expect(caseRows.value[1].bestVersionIndex).toBe(1);
expect(caseRows.value[1].cells.map((c) => c.score)).toEqual([0.5, 0.9]);
});
it('flags a dataset mismatch and null-fills missing cells when case counts diverge', async () => {
// helpfulness is a 1–5 AI-judge metric → normalized /5 (4 → 0.8 etc.).
seed([
caseRecord({ id: 'a0', testRunId: 'run-a', runIndex: 0, metrics: { helpfulness: 3.5 } }),
caseRecord({ id: 'a1', testRunId: 'run-a', runIndex: 1, metrics: { helpfulness: 4 } }),
caseRecord({ id: 'b0', testRunId: 'run-b', runIndex: 0, metrics: { helpfulness: 3 } }),
]);
const { caseRows, mismatch } = useCompareCases(
ref(detailWith(['run-a', 'run-b'])),
ref('wf-1'),
);
await nextTick();
expect(mismatch.value.hasMismatch).toBe(true);
expect(mismatch.value.counts).toEqual([2, 1]);
// run-b has no case #1 → its cell is null-filled, run-a keeps its value.
expect(caseRows.value[1].cells[0].score).toBe(0.8);
expect(caseRows.value[1].cells[1].testCaseId).toBeNull();
expect(caseRows.value[1].cells[1].score).toBeNull();
});
it('fans out fetchTestCaseExecutions once per run and toggles loading', async () => {
const fetchSpy = vi.fn(async ({ runId }: { workflowId: string; runId: string }) => {
store.$patch((state) => {
state.testCaseExecutionsById = {
...state.testCaseExecutionsById,
[`${runId}-0`]: caseRecord({ id: `${runId}-0`, testRunId: runId, runIndex: 0 }),
};
});
return [];
});
store.fetchTestCaseExecutions = fetchSpy as unknown as typeof store.fetchTestCaseExecutions;
const { loading, casesLoaded, caseRows, load } = useCompareCases(
ref(detailWith(['run-a', 'run-b'])),
ref('wf-1'),
);
await load();
expect(fetchSpy).toHaveBeenCalledWith({ workflowId: 'wf-1', runId: 'run-a' });
expect(fetchSpy).toHaveBeenCalledWith({ workflowId: 'wf-1', runId: 'run-b' });
expect(loading.value).toBe(false);
expect(casesLoaded.value).toBe(true);
expect(caseRows.value).toHaveLength(1);
});
it('flags casesError when a run fetch rejects (not a real mismatch)', async () => {
store.fetchTestCaseExecutions = vi.fn(async ({ runId }: { runId: string }) => {
if (runId === 'run-b') throw new Error('network');
return [];
}) as unknown as typeof store.fetchTestCaseExecutions;
const { casesError, load } = useCompareCases(ref(detailWith(['run-a', 'run-b'])), ref('wf-1'));
await load();
expect(casesError.value).toBe(true);
});
it('excludes predefined operational metrics from the per-case score', async () => {
seed([
caseRecord({
id: 'a0',
testRunId: 'run-a',
runIndex: 0,
metrics: { helpfulness: 4, totalTokens: 1, executionTime: 0.4 },
}),
]);
const { caseRows } = useCompareCases(ref(detailWith(['run-a'])), ref('wf-1'));
await nextTick();
// score is the mean of normalized score metrics → just helpfulness (4/5 = 0.8);
// tokens/execution time are operational and excluded.
expect(caseRows.value[0].cells[0].score).toBe(0.8);
});
it('aligns by runIndex so a version missing a middle case does not shift later cases', async () => {
seed([
caseRecord({ id: 'a0', testRunId: 'run-a', runIndex: 0, metrics: { helpfulness: 5 } }),
caseRecord({ id: 'a1', testRunId: 'run-a', runIndex: 1, metrics: { helpfulness: 4 } }),
caseRecord({ id: 'a2', testRunId: 'run-a', runIndex: 2, metrics: { helpfulness: 3 } }),
// run-b is missing the middle case (runIndex 1).
caseRecord({ id: 'b0', testRunId: 'run-b', runIndex: 0, metrics: { helpfulness: 5 } }),
caseRecord({ id: 'b2', testRunId: 'run-b', runIndex: 2, metrics: { helpfulness: 2 } }),
]);
const { caseRows } = useCompareCases(ref(detailWith(['run-a', 'run-b'])), ref('wf-1'));
await nextTick();
expect(caseRows.value).toHaveLength(3);
// runIndex 1: run-b has no case → null cell, not run-b's next case shifted up.
expect(caseRows.value[1].cells[0].testCaseId).toBe('a1');
expect(caseRows.value[1].cells[1].testCaseId).toBeNull();
// runIndex 2: b2 stays paired with a2 (same seeded case), not with a1.
expect(caseRows.value[2].cells[0].testCaseId).toBe('a2');
expect(caseRows.value[2].cells[1].testCaseId).toBe('b2');
});
it('resolves to a loaded (not stuck) state for an empty collection', async () => {
const { casesLoaded, loading, caseRows } = useCompareCases(ref(detailWith([])), ref('wf-1'));
await nextTick();
// The watcher must still invoke load() for a detail with no runs so its
// empty-run completion branch runs, rather than sitting in loading forever.
expect(casesLoaded.value).toBe(true);
expect(loading.value).toBe(false);
expect(caseRows.value).toEqual([]);
});
});
@@ -0,0 +1,195 @@
import orderBy from 'lodash/orderBy';
import type { JsonObject } from 'n8n-workflow';
import { computed, ref, watch, type Ref } from 'vue';
import type { TestCaseExecutionRecord } from '../evaluation.api';
import { useEvaluationStore } from '../evaluation.store';
import type { EvaluationCollectionDetail } from '../evalCollections.types';
import { averageNormalizedScore, indexOfMax, stringifyValue } from '../evaluation.utils';
// One version's execution of a single aligned test case. `null`-valued fields
// mark a case that this version's run didn't cover (dataset drift).
export interface CompareCaseCell {
versionIndex: number;
testCaseId: string | null;
inputs: JsonObject | undefined;
outputs: JsonObject | undefined;
metrics: Record<string, number> | undefined;
// Mean of the case's score-shaped ([0, 1], non-predefined) metrics, or null
// when the version skipped this case or reported no score-shaped metric.
score: number | null;
}
export interface CompareCaseRow {
index: number;
displayIndex: number;
inputPreview: string;
cells: CompareCaseCell[];
// Version index with the highest case score, or null if none scored.
bestVersionIndex: number | null;
}
export interface DatasetMismatch {
hasMismatch: boolean;
// Case count per version, aligned to run order.
counts: number[];
maxCount: number;
}
// Compact one-line preview of a case's inputs for the table's first column.
function inputPreview(inputs: JsonObject | undefined): string {
if (!inputs) return '';
return Object.values(inputs)
.map((value) => stringifyValue(value))
.filter((text) => text.length > 0)
.join(' · ');
}
/**
* Loads per-case executions for every run in a collection and aligns them into
* one row per test case across versions.
*
* There is no collection-level per-case endpoint and no case id shared across
* runs, so this fans out `fetchTestCaseExecutions` per run and aligns cells by
* `runIndex` (the seeded per-case sequence) — a version missing a case leaves a
* null cell rather than shifting later cases into the wrong row. Divergent case
* counts surface as a `mismatch` rather than silently misaligning rows.
*/
export function useCompareCases(
detail: Ref<EvaluationCollectionDetail | null>,
workflowId: Ref<string>,
) {
const evaluationStore = useEvaluationStore();
const loading = ref(false);
// True once the current run set's per-case fetches have completed at least
// once. Downstream gates (mismatch banner, telemetry) use this rather than
// `!loading` so they can't act on the empty window before the first load or
// on a superseded load's transient `loading = false`.
const casesLoaded = ref(false);
// True when any run's per-case fetch failed. Distinguishes a transient
// failure from a real dataset mismatch — a failed run also comes back with
// zero cases, which would otherwise read as "diverging case counts".
const casesError = ref(false);
// Monotonic token so a slow load for a previous collection can't flip state
// out from under the collection the user has since switched to.
let loadToken = 0;
async function load() {
const runs = detail.value?.runs ?? [];
const token = ++loadToken;
if (runs.length === 0) {
loading.value = false;
casesError.value = false;
casesLoaded.value = true;
return;
}
loading.value = true;
casesLoaded.value = false;
casesError.value = false;
try {
const results = await Promise.allSettled(
runs.map(
async (run) =>
await evaluationStore.fetchTestCaseExecutions({
workflowId: workflowId.value,
runId: run.testRunId,
}),
),
);
// A newer load for a different run set has taken over — don't clobber it.
if (token !== loadToken) return;
casesError.value = results.some((result) => result.status === 'rejected');
casesLoaded.value = true;
} finally {
if (token === loadToken) loading.value = false;
}
}
// Per-run, sorted case lists. Bucket the shared (app-global, poll-mutated)
// store map by `testRunId` in a single pass instead of re-scanning it per
// run, then sort each run's bucket by the same [runIndex, runAt] ordering
// the run-detail view uses so aligned positions map to the same seeded case.
const casesByVersion = computed<TestCaseExecutionRecord[][]>(() => {
const runs = detail.value?.runs ?? [];
const byRunId = new Map<string, TestCaseExecutionRecord[]>(
runs.map((run) => [run.testRunId, []]),
);
for (const record of Object.values(evaluationStore.testCaseExecutionsById)) {
const bucket = record.testRunId ? byRunId.get(record.testRunId) : undefined;
if (bucket) bucket.push(record);
}
return runs.map((run) =>
orderBy(
byRunId.get(run.testRunId) ?? [],
[(record) => record.runIndex ?? Number.MAX_SAFE_INTEGER, (record) => record.runAt ?? ''],
['asc', 'asc'],
),
);
});
const mismatch = computed<DatasetMismatch>(() => {
const counts = casesByVersion.value.map((cases) => cases.length);
const maxCount = counts.length ? Math.max(...counts) : 0;
return {
counts,
maxCount,
hasMismatch: counts.some((count) => count !== maxCount),
};
});
const caseRows = computed<CompareCaseRow[]>(() => {
// Align cells by `runIndex` (the seeded per-case sequence), not list
// position: a version missing a case in the middle must leave a null cell
// in that row rather than shift every later case up and pair unrelated
// inputs. Fall back to list position only when a record has no runIndex.
const byIndex = casesByVersion.value.map((cases) => {
const map = new Map<number, TestCaseExecutionRecord>();
cases.forEach((record, position) => map.set(record.runIndex ?? position, record));
return map;
});
const allIndices = [...new Set(byIndex.flatMap((map) => [...map.keys()]))].sort(
(a, b) => a - b,
);
return allIndices.map((runIndex, rowIndex) => {
const cells: CompareCaseCell[] = byIndex.map((map, versionIndex) => {
const record = map.get(runIndex);
return {
versionIndex,
testCaseId: record?.id ?? null,
inputs: record?.inputs,
outputs: record?.outputs,
metrics: record?.metrics,
score: averageNormalizedScore(record?.metrics),
};
});
const firstWithInputs = cells.find((cell) => cell.inputs !== undefined);
return {
index: rowIndex,
displayIndex: rowIndex + 1,
inputPreview: inputPreview(firstWithInputs?.inputs),
cells,
bestVersionIndex: indexOfMax(cells.map((cell) => cell.score)),
};
});
});
// Refetch whenever the run set changes (collection switch reuses the view).
// The key is `null` while the detail hasn't loaded and an empty string once
// it has but with no runs — distinguishing them so an empty collection still
// runs `load()` (which resolves its own empty-run completion state) instead
// of sitting in the initial loading state forever.
watch(
() => (detail.value ? (detail.value.runs ?? []).map((run) => run.testRunId).join(',') : null),
async (key) => {
if (key !== null) await load();
},
{ immediate: true },
);
return { caseRows, mismatch, loading, casesLoaded, casesError, load };
}
@@ -62,7 +62,7 @@ describe('useCompareData', () => {
expect(versions[1].label).toBe('abcdef1');
});
it('charts only score-shaped metrics and picks the best version per metric', () => {
it('charts score metrics normalized to [0,1] and picks the best version per metric', () => {
const source = ref(
detail([
{
@@ -72,7 +72,8 @@ describe('useCompareData', () => {
runAt: null,
completedAt: null,
avgScore: 0.7,
metrics: { helpfulness: 0.7, totalTokens: 1200 },
// helpfulness is a 1–5 AI-judge metric → normalized /5.
metrics: { helpfulness: 3.5, totalTokens: 1200 },
},
{
testRunId: 'run-b',
@@ -81,14 +82,14 @@ describe('useCompareData', () => {
runAt: null,
completedAt: null,
avgScore: 0.9,
metrics: { helpfulness: 0.9, totalTokens: 1500 },
metrics: { helpfulness: 4.5, totalTokens: 1500 },
},
]),
);
const { compareData } = useCompareData(source);
const groups = compareData.value!.metricGroups;
// totalTokens (out of [0,1]) is excluded — only helpfulness is charted.
// totalTokens is an operational metric → excluded; helpfulness charts at /5.
expect(groups).toHaveLength(1);
expect(groups[0].key).toBe('helpfulness');
expect(groups[0].values).toEqual([0.7, 0.9]);
@@ -108,7 +109,7 @@ describe('useCompareData', () => {
avgScore: 0.7,
// executionTime happens to be in [0, 1] here, but it's an absolute
// operational metric — it must not be charted as a score.
metrics: { helpfulness: 0.7, executionTime: 0.4, totalTokens: 1 },
metrics: { helpfulness: 3.5, executionTime: 0.4, totalTokens: 1 },
},
]),
);
@@ -128,7 +129,8 @@ describe('useCompareData', () => {
runAt: null,
completedAt: null,
avgScore: 0.6,
metrics: { correctness: 0.6 },
// correctness/helpfulness are 1–5 AI-judge metrics → normalized /5.
metrics: { correctness: 3 },
},
{
testRunId: 'run-b',
@@ -137,7 +139,7 @@ describe('useCompareData', () => {
runAt: null,
completedAt: null,
avgScore: 0.8,
metrics: { correctness: 0.8, helpfulness: 0.5 },
metrics: { correctness: 4, helpfulness: 2.5 },
},
]),
);
@@ -3,7 +3,7 @@ import { computed, type Ref } from 'vue';
import { useI18n } from '@n8n/i18n';
import type { EvalCollectionRunStatus, EvaluationCollectionDetail } from '../evalCollections.types';
import { buildScoreShapedMetricGroups, formatMetricLabel } from '../evaluation.utils';
import { buildScoreShapedMetricGroups, formatMetricLabel, indexOfMax } from '../evaluation.utils';
import { versionLetter } from '../components/shared/versionPalette';
// One column in the compare view: a single run positioned by `index`, which
@@ -35,20 +35,6 @@ export interface CompareData {
bestVersionIndex: number | null;
}
// Index of the max value in `values`, ignoring nulls. Ties resolve to the
// first (left-most) version, matching the letter order users read.
function indexOfMax(values: Array<number | null>): number | null {
let best: number | null = null;
let bestValue = -Infinity;
values.forEach((value, index) => {
if (value !== null && value > bestValue) {
bestValue = value;
best = index;
}
});
return best;
}
/**
* Shapes a collection's aggregate detail into the compare view's model:
* one `CompareVersion` per run (in stored order) and one `CompareMetricGroup`
@@ -0,0 +1,41 @@
import { describe, expect, it, vi } from 'vitest';
import { useEvalCollectionsFlag } from './useEvalCollectionsFlag';
const settingsState = { collectionsEnabled: false };
const posthogState = { enabled: false };
vi.mock('@/app/stores/settings.store', () => ({
useSettingsStore: () => ({
settings: { evaluation: { collectionsEnabled: settingsState.collectionsEnabled } },
}),
}));
vi.mock('@/app/stores/posthog.store', () => ({
usePostHog: () => ({ isFeatureEnabled: () => posthogState.enabled }),
}));
describe('useEvalCollectionsFlag', () => {
it('is enabled via the backend operator override even when PostHog is off', () => {
// The telemetry-off case: the in-browser PostHog client never initializes,
// so the flag must come from the settings-provided override.
settingsState.collectionsEnabled = true;
posthogState.enabled = false;
expect(useEvalCollectionsFlag().value).toBe(true);
});
it('is enabled via the PostHog cohort flag when the override is off', () => {
settingsState.collectionsEnabled = false;
posthogState.enabled = true;
expect(useEvalCollectionsFlag().value).toBe(true);
});
it('is disabled when neither signal is set', () => {
settingsState.collectionsEnabled = false;
posthogState.enabled = false;
expect(useEvalCollectionsFlag().value).toBe(false);
});
});
@@ -2,20 +2,28 @@ import { EVAL_COLLECTIONS_FLAG } from '@n8n/api-types';
import { computed } from 'vue';
import { usePostHog } from '@/app/stores/posthog.store';
import { useSettingsStore } from '@/app/stores/settings.store';
/**
* Frontend gate for the eval-collections feature surface. Mirrors the
* `084_eval_collections` PostHog rollout flag that the backend consults to
* 404 the controller routes. The env override
* `N8N_EVAL_COLLECTIONS_ENABLED=true` flips PostHog to "enabled for every
* user on the running main" — useful for local + QA — without round-tripping
* the cohort layer.
* Frontend gate for the eval-collections feature surface, matching the
* `084_eval_collections` flag the backend consults to 404 the controller
* routes. It combines two independent signals:
*
* Coerces PostHog's `boolean | undefined` return to a strict boolean so
* `v-if="isEvalCollectionsEnabled"` is never undefined-flickering during
* the initial flag-fetch frame.
* - `settings.evaluation.collectionsEnabled` — the backend-provided operator
* override (`N8N_EVAL_COLLECTIONS_ENABLED`). Delivered in the settings
* payload, so it works even when the in-browser PostHog client never
* initializes (telemetry off), where the flag would otherwise stay false.
* - the PostHog client flag — carries per-cohort rollout when telemetry is on.
*
* Coerces to a strict boolean so `v-if="isEvalCollectionsEnabled"` never
* undefined-flickers during the initial flag-fetch frame.
*/
export const useEvalCollectionsFlag = () => {
const postHog = usePostHog();
return computed(() => postHog.isFeatureEnabled(EVAL_COLLECTIONS_FLAG) === true);
const settingsStore = useSettingsStore();
return computed(
() =>
settingsStore.settings.evaluation?.collectionsEnabled === true ||
postHog.isFeatureEnabled(EVAL_COLLECTIONS_FLAG) === true,
);
};
@@ -16,6 +16,7 @@ import {
getDefaultOrderedColumns,
getDeltaTone,
getMetricCategory,
getMetricDescriptionKey,
getTestCasesColumns,
getTestTableHeaders,
getUserDefinedMetricNames,
@@ -1451,6 +1452,28 @@ describe('utils', () => {
});
});
describe('getMetricDescriptionKey', () => {
it('returns an i18n key for each built-in metric', () => {
expect(getMetricDescriptionKey('correctness')).toBe(
'evaluation.metric.description.correctness',
);
expect(getMetricDescriptionKey('helpfulness')).toBe(
'evaluation.metric.description.helpfulness',
);
expect(getMetricDescriptionKey('stringSimilarity')).toBe(
'evaluation.metric.description.stringSimilarity',
);
expect(getMetricDescriptionKey('categorization')).toBe(
'evaluation.metric.description.categorization',
);
expect(getMetricDescriptionKey('toolsUsed')).toBe('evaluation.metric.description.toolsUsed');
});
it('returns null for custom/unknown metrics', () => {
expect(getMetricDescriptionKey('myCustomMetric')).toBeNull();
expect(getMetricDescriptionKey(undefined)).toBeNull();
});
});
describe('extractAnswerText', () => {
it('returns null as empty string', () => {
expect(extractAnswerText(null)).toBe('');
@@ -1,4 +1,10 @@
import startCase from 'lodash/startCase';
import {
normalizeMetricScore,
ONE_TO_FIVE_METRIC_KEYS,
RESERVED_METRIC_KEYS,
} from '@n8n/api-types';
import type { BaseTextKey } from '@n8n/i18n';
import type { JsonValue } from 'n8n-workflow';
import type { IconName } from '@n8n/design-system/components/N8nIcon/icons';
import type { IExecutionResponse } from '@/features/execution/executions/executions.types';
@@ -81,12 +87,7 @@ export type MetricSource = {
export const SHORT_TABLE_CELL_MIN_WIDTH = 125;
const LONG_TABLE_CELL_MIN_WIDTH = 250;
const PREDEFINED_METRIC_KEYS: ReadonlySet<string> = new Set([
'promptTokens',
'completionTokens',
'totalTokens',
'executionTime',
]);
const PREDEFINED_METRIC_KEYS: ReadonlySet<string> = new Set(RESERVED_METRIC_KEYS);
// Excludes predefined keys (token counts, execution time) emitted by every run.
export function getUserDefinedMetricNames(
@@ -112,33 +113,48 @@ export function normalizeMetricValue(value: number | undefined): number | undefi
return value;
}
// A metric value is "score-shaped" when it lands in [0, 1] — the range the
// collection cards chart and average as a percentage. Absolute counts that
// commonly share the metrics map (tokens, latency_ms) fall outside and are
// excluded so a mini bar chart (clamped to max=1) doesn't render a bogus
// maxed-out bar and an avg doesn't blow up.
export function isScoreShapedMetric(value: unknown): value is number {
return typeof value === 'number' && value >= 0 && value <= 1;
// Index of the max value, ignoring nulls; ties resolve to the first (left-most)
// entry, matching the version letter order users read. Returns null if every
// value is null.
export function indexOfMax(values: Array<number | null>): number | null {
let best: number | null = null;
let bestValue = -Infinity;
values.forEach((value, index) => {
if (value !== null && value > bestValue) {
bestValue = value;
best = index;
}
});
return best;
}
// A run set is "running" while any run is still queued or executing, else
// "done". Shared by the collection card and the compare header so the two
// surfaces can't disagree; callers that also have a not-yet-loaded state keep
// their own `null` guard around this.
// Overall status of a run set: "running" while any run is still queued or
// executing, "error" when every run failed or was cancelled (so an all-failed
// collection doesn't read as a green "done"), otherwise "done". Shared by the
// collection card and the compare header so the two surfaces can't disagree;
// callers that also have a not-yet-loaded state keep their own `null` guard.
export function deriveRunsStatus(
runs: Array<{ status: EvalCollectionRunStatus }>,
): 'running' | 'done' {
return runs.some((run) => run.status === 'new' || run.status === 'running') ? 'running' : 'done';
): 'running' | 'done' | 'error' {
if (runs.some((run) => run.status === 'new' || run.status === 'running')) return 'running';
if (
runs.length > 0 &&
runs.every((run) => run.status === 'error' || run.status === 'cancelled')
) {
return 'error';
}
return 'done';
}
// Reduce per-run aggregate metrics to the score-shaped ([0, 1]) metrics that
// both the collection-card preview and the compare hero chart render. Returns
// one entry per metric (first-seen order) with a value per run aligned by
// index — `null` where a run lacks the metric, so a skipped metric never
// shifts later versions out of their color/letter slot. A metric is included
// only if it's score-shaped across every run that reported it, since the bar
// charts clamp to max=1 and an absolute count (tokens, latency) would render a
// meaningless maxed-out bar.
// Reduce per-run aggregate metrics to the score metrics that both the
// collection-card preview and the compare hero chart render, each normalized to
// [0, 1] by its scale (AI-judge metrics are 1–5 → /5; see `normalizeMetricScore`).
// Returns one entry per metric (first-seen order) with a value per run aligned by
// index — `null` where a run lacks the metric, so a skipped metric never shifts
// later versions out of their color/letter slot. A metric is included only if
// every run that reported it yields a score (operational counts like tokens and
// latency normalize to `null` and are dropped, since the bar charts clamp to
// max=1 and an absolute count would render a meaningless maxed-out bar).
export function buildScoreShapedMetricGroups(
runs: Array<{ metrics: Record<string, number> | null }>,
): Array<{ key: string; values: Array<number | null> }> {
@@ -146,32 +162,45 @@ export function buildScoreShapedMetricGroups(
const seen = new Set<string>();
for (const run of runs) {
for (const key of Object.keys(run.metrics ?? {})) {
// Skip predefined operational metrics (token counts, execution time) —
// they're absolute values, not scores, and would chart as a bogus
// percentage on the rare run where they land in [0, 1]. Matches the
// exclusion in `getUserDefinedMetricNames`.
if (PREDEFINED_METRIC_KEYS.has(key) || seen.has(key)) continue;
if (seen.has(key)) continue;
seen.add(key);
orderedKeys.push(key);
}
}
const scoreShapedKeys = orderedKeys.filter((key) =>
runs.every((run) => {
const value = run.metrics?.[key];
return value === undefined || isScoreShapedMetric(value);
}),
const scoreKeys = orderedKeys.filter(
(key) =>
runs.some((run) => run.metrics?.[key] !== undefined) &&
runs.every((run) => {
const value = run.metrics?.[key];
return value === undefined || normalizeMetricScore(key, value) !== null;
}),
);
return scoreShapedKeys.map((key) => ({
return scoreKeys.map((key) => ({
key,
values: runs.map((run) => {
const value = run.metrics?.[key];
return typeof value === 'number' ? value : null;
return typeof value === 'number' ? normalizeMetricScore(key, value) : null;
}),
}));
}
// Mean of a metrics map's score values, each normalized to [0, 1] by its scale.
// Returns null when no metric qualifies (only operational counts, or an empty
// map). Single definition so the cards, cases table, and hero chart can't
// disagree on what a case/run scored.
export function averageNormalizedScore(
metrics: Record<string, number> | null | undefined,
): number | null {
if (!metrics) return null;
const values = Object.entries(metrics)
.map(([key, value]) => normalizeMetricScore(key, value))
.filter((value): value is number => value !== null);
if (values.length === 0) return null;
return values.reduce((sum, value) => sum + value, 0) / values.length;
}
export function computeDelta(
current: number | undefined,
previous: number | undefined,
@@ -259,10 +288,10 @@ export function formatMetricLabel(name: string): string {
// `correctness` + `helpfulness` collapse into 'aiBased' (both LLM-as-judge).
export function getMetricCategory(metric: string | undefined): MetricCategory {
if (metric !== undefined && (ONE_TO_FIVE_METRIC_KEYS as readonly string[]).includes(metric)) {
return 'aiBased';
}
switch (metric) {
case 'correctness':
case 'helpfulness':
return 'aiBased';
case 'stringSimilarity':
return 'stringSimilarity';
case 'categorization':
@@ -274,6 +303,23 @@ export function getMetricCategory(metric: string | undefined): MetricCategory {
}
}
// Short "what this measures" copy for the built-in metrics, mirrored from the
// Evaluation node's metric options. Custom/unknown metrics have no canned
// description (the UI just shows the name).
const METRIC_DESCRIPTION_KEYS: Partial<Record<string, BaseTextKey>> = {
correctness: 'evaluation.metric.description.correctness',
helpfulness: 'evaluation.metric.description.helpfulness',
stringSimilarity: 'evaluation.metric.description.stringSimilarity',
categorization: 'evaluation.metric.description.categorization',
toolsUsed: 'evaluation.metric.description.toolsUsed',
};
// i18n key for a metric's description, or null for custom/unknown metrics.
export function getMetricDescriptionKey(metric: string | undefined): BaseTextKey | null {
if (metric === undefined) return null;
return METRIC_DESCRIPTION_KEYS[metric] ?? null;
}
function formatScoreNumerator(value: number): string {
const rounded = Math.round(value * 10) / 10;
return Number.isInteger(rounded) ? `${rounded}` : rounded.toFixed(1);
@@ -6,9 +6,15 @@ import { createComponentRenderer } from '@/__tests__/render';
import { VIEWS } from '@/app/constants';
import { useEvalCollectionsStore } from '../evalCollections.store';
import { useEvaluationStore } from '../evaluation.store';
import type { EvaluationCollectionDetail } from '../evalCollections.types';
import CompareCollectionView from './CompareCollectionView.vue';
const track = vi.fn();
vi.mock('@/app/composables/useTelemetry', () => ({
useTelemetry: () => ({ track }),
}));
const routerReplace = vi.fn();
vi.mock('vue-router', async (importOriginal) => ({
...(await importOriginal<typeof import('vue-router')>()),
@@ -82,18 +88,29 @@ const renderComponent = createComponentRenderer(CompareCollectionView, {
describe('CompareCollectionView', () => {
let store: ReturnType<typeof useEvalCollectionsStore>;
let evaluationStore: ReturnType<typeof useEvaluationStore>;
beforeEach(() => {
flagState.enabled = true;
routerReplace.mockClear();
track.mockClear();
store = useEvalCollectionsStore();
evaluationStore = useEvaluationStore();
store.stopPolling = vi.fn() as unknown as typeof store.stopPolling;
// Per-case fetch is stubbed to a no-op by default so useCompareCases
// doesn't hit the network; tests that assert on cases seed the map.
evaluationStore.fetchTestCaseExecutions = vi.fn(
async () => [],
) as unknown as typeof evaluationStore.fetchTestCaseExecutions;
// Hard-replace the shared testing pinia's maps so a prior test's cached
// detail doesn't leak in (object-form `$patch` deep-merges stale keys).
store.$patch((state) => {
state.collectionDetailById = {};
state.loadingDetail = {};
});
evaluationStore.$patch((state) => {
state.testCaseExecutionsById = {};
});
});
it('redirects to the evaluations list when the flag is off', async () => {
@@ -139,6 +156,33 @@ describe('CompareCollectionView', () => {
expect(container.textContent).toContain('Tone tuning experiment');
});
it('renders the compare tabs and fires the compare-opened event once data loads', async () => {
store.fetchCollectionDetail = vi.fn(async () => {
store.$patch({ collectionDetailById: { 'col-1': DETAIL } });
return DETAIL;
}) as unknown as typeof store.fetchCollectionDetail;
const { container } = renderComponent();
await waitFor(() =>
expect(container.querySelector('[data-test-id="compare-tabs"]')).not.toBeNull(),
);
await waitFor(() =>
expect(track).toHaveBeenCalledWith(
'Eval collection compared opened',
expect.objectContaining({
workflow_id: 'wf-1',
collection_id: 'col-1',
version_count: 2,
}),
),
);
// fired exactly once for the collection
expect(track.mock.calls.filter((c) => c[0] === 'Eval collection compared opened')).toHaveLength(
1,
);
});
it('shows the not-found state when the collection has no detail', async () => {
store.fetchCollectionDetail = vi.fn(
async () => undefined,
@@ -6,14 +6,19 @@ import { useRouter } from 'vue-router';
import { VIEWS } from '@/app/constants';
import { useToast } from '@/app/composables/useToast';
import { useTelemetry } from '@/app/composables/useTelemetry';
import { usePostHog } from '@/app/stores/posthog.store';
import CompareHeader from '../components/Compare/CompareHeader.vue';
import ScoreChart from '../components/Compare/ScoreChart.vue';
import AiInsightsCard from '../components/Compare/AiInsightsCard.vue';
import CompareTabs from '../components/Compare/CompareTabs.vue';
import DatasetMismatchBanner from '../components/Compare/DatasetMismatchBanner.vue';
import { useCompareData } from '../composables/useCompareData';
import { useCompareCases } from '../composables/useCompareCases';
import { useEvalCollectionsFlag } from '../composables/useEvalCollectionsFlag';
import { useEvalCollectionsStore } from '../evalCollections.store';
import { useEvaluationStore } from '../evaluation.store';
const props = defineProps<{
workflowId: string;
@@ -23,12 +28,60 @@ const props = defineProps<{
const i18n = useI18n();
const router = useRouter();
const toast = useToast();
const telemetry = useTelemetry();
const store = useEvalCollectionsStore();
const evaluationStore = useEvaluationStore();
const postHog = usePostHog();
const isEvalCollectionsEnabled = useEvalCollectionsFlag();
const detail = computed(() => store.getDetail(props.collectionId));
// metric name → its custom LLM-judge prompt (the specific criteria the user
// configured), sourced from the collection's evaluation config. Run metrics are
// keyed by the metric's `name` (see the workflow compiler), so the map keys line
// up with the compare view's metric keys. Empty until the config resolves.
const metricPrompts = computed<Record<string, string>>(() => {
const configId = detail.value?.evaluationConfigId;
if (!configId) return {};
const config = (evaluationStore.evaluationConfigsByWorkflowId[props.workflowId] ?? []).find(
(candidate) => candidate.id === configId,
);
if (!config) return {};
const prompts: Record<string, string> = {};
for (const metric of config.metrics) {
if (metric.type === 'llm_judge' && metric.config.prompt) {
prompts[metric.name] = metric.config.prompt;
}
}
return prompts;
});
const { compareData } = useCompareData(detail);
const workflowIdRef = computed(() => props.workflowId);
const {
caseRows,
mismatch,
loading: casesLoading,
casesLoaded,
casesError,
} = useCompareCases(detail, workflowIdRef);
// Fire the compare-opened event once per collection, after both the versions
// and the per-case data have resolved so `case_count` is accurate.
const tracked = ref(false);
watch(
() => compareData.value !== null && casesLoaded.value,
(ready) => {
if (!ready || tracked.value) return;
tracked.value = true;
telemetry.track('Eval collection compared opened', {
workflow_id: props.workflowId,
collection_id: props.collectionId,
version_count: compareData.value?.versions.length ?? 0,
case_count: mismatch.value.maxCount,
});
},
{ immediate: true },
);
const loading = computed(() => store.loadingDetail[props.collectionId] ?? false);
// Set only when the collection is genuinely gone (404), so a deleted collection
@@ -49,13 +102,31 @@ function isNotFoundError(error: unknown): boolean {
);
}
// Tracks unmount so a fetch that resolves after the user leaves can tear down
// the poll it armed instead of letting it outlive the view.
let unmounted = false;
async function load(workflowId: string, collectionId: string) {
notFound.value = false;
try {
await store.fetchCollectionDetail(workflowId, collectionId);
// Best-effort: metric criteria come from the eval config. A failure here
// just means the compare view shows metric names without their criteria.
await evaluationStore.fetchEvaluationConfigs(workflowId).catch(() => null);
// If we left or switched collections mid-fetch, `fetchCollectionDetail`
// may have just (re)armed polling for a collection we're no longer
// showing — stop it so the timer doesn't outlive the view.
if (unmounted || collectionId !== props.collectionId) {
store.stopPolling(collectionId);
}
} catch (error) {
toast.showError(error, i18n.baseText('evaluation.compare.errors.loadFailed'));
notFound.value = isNotFoundError(error);
// A 404 already shows the not-found state; a toast on top would be a
// second, contradictory signal. Only toast transient (non-404) failures.
if (isNotFoundError(error)) {
notFound.value = true;
} else {
toast.showError(error, i18n.baseText('evaluation.compare.errors.loadFailed'));
}
}
}
@@ -84,11 +155,13 @@ watch(
[() => props.workflowId, () => props.collectionId],
([, collectionId], [, prevCollectionId]) => {
store.stopPolling(prevCollectionId);
tracked.value = false;
void load(props.workflowId, collectionId);
},
);
onBeforeUnmount(() => {
unmounted = true;
store.stopPolling(props.collectionId);
});
</script>
@@ -120,16 +193,33 @@ onBeforeUnmount(() => {
</div>
<template v-else-if="compareData">
<DatasetMismatchBanner
v-if="casesLoaded && !casesError && mismatch.hasMismatch"
:mismatch="mismatch"
/>
<CompareHeader
:collection-name="detail?.name ?? ''"
:versions="compareData.versions"
:best-version-index="compareData.bestVersionIndex"
/>
<ScoreChart :metric-groups="compareData.metricGroups" :versions="compareData.versions" />
<ScoreChart
:metric-groups="compareData.metricGroups"
:versions="compareData.versions"
:metric-prompts="metricPrompts"
/>
<!-- Key by collection so navigating between compare views remounts the
card and re-runs its fetch-on-mount, rather than relying on the
surrounding v-if to cycle through null. -->
<AiInsightsCard :key="collectionId" :workflow-id="workflowId" :collection-id="collectionId" />
<CompareTabs
:versions="compareData.versions"
:metric-groups="compareData.metricGroups"
:case-rows="caseRows"
:cases-loading="casesLoading"
:cases-error="casesError"
:workflow-id="workflowId"
:metric-prompts="metricPrompts"
/>
</template>
</div>
</template>
@@ -1,5 +1,5 @@
<script setup lang="ts">
import { N8nBadge, N8nButton, N8nIcon, N8nText } from '@n8n/design-system';
import { N8nBadge, N8nButton, N8nIcon, N8nPagination, N8nText } from '@n8n/design-system';
import { useI18n } from '@n8n/i18n';
import { computed, onBeforeUnmount, onMounted, ref, watch } from 'vue';
@@ -13,6 +13,9 @@ import { useEvaluationStore } from '../evaluation.store';
const props = defineProps<{
workflowId: string;
// When set, the collections list paginates to this many cards per page (used
// when stacked with the config-eval hub). Omitted → the full list renders.
pageSize?: number;
}>();
const i18n = useI18n();
@@ -24,6 +27,28 @@ const wizardOpen = ref(false);
const collections = computed(() => store.getCollections(props.workflowId));
const collectionsPage = ref(1);
const showCollectionsPagination = computed(
() => !!props.pageSize && collections.value.length > props.pageSize,
);
const pagedCollections = computed(() => {
if (!props.pageSize) return collections.value;
const start = (collectionsPage.value - 1) * props.pageSize;
return collections.value.slice(start, start + props.pageSize);
});
// Clamp the page if the collection set shrinks (delete) so we don't strand the
// user on an empty page past the end.
watch(
() => collections.value.length,
() => {
const maxPage = props.pageSize
? Math.max(1, Math.ceil(collections.value.length / props.pageSize))
: 1;
if (collectionsPage.value > maxPage) collectionsPage.value = maxPage;
},
);
const ungroupedRuns = computed(() => {
const all = evaluationStore.testRunsByWorkflowId[props.workflowId] ?? [];
return all
@@ -95,6 +120,10 @@ watch(
() => props.workflowId,
async (next, prev) => {
if (next === prev) return;
// Reset to the first page: the length-based clamp below won't fire when the
// new workflow happens to have the same cached collection count, which would
// otherwise strand the user on the previous workflow's page.
collectionsPage.value = 1;
stopAllPolling();
await loadForWorkflow(next);
},
@@ -106,94 +135,116 @@ onBeforeUnmount(() => {
</script>
<template>
<div :class="$style.view" data-test-id="eval-collections-list-view">
<header :class="$style.header">
<div :class="$style.headerText">
<N8nText tag="h2" size="xlarge" bold>
{{ i18n.baseText('evaluation.collections.title') }}
</N8nText>
<N8nText tag="p" size="small" color="text-base">
{{ i18n.baseText('evaluation.collections.subtitle') }}
</N8nText>
</div>
<N8nButton
variant="solid"
icon="plus"
:label="i18n.baseText('evaluation.collections.newCollection')"
data-test-id="eval-collections-new-button"
@click="onOpenWizard"
/>
</header>
<section :class="$style.section">
<div :class="$style.sectionHeader">
<N8nText tag="h3" size="medium" bold>
{{ i18n.baseText('evaluation.collections.section.collections') }}
</N8nText>
<N8nBadge theme="tertiary" size="small">{{ collections.length }}</N8nBadge>
</div>
<div v-if="collections.length === 0" :class="$style.emptyHint">
<N8nIcon icon="layers" size="medium" />
<div :class="$style.emptyHintBody">
<N8nText size="small" bold>
{{ i18n.baseText('evaluation.collections.empty.title') }}
<div :class="$style.viewWrapper">
<div :class="$style.view" data-test-id="eval-collections-list-view">
<header :class="$style.header">
<div :class="$style.headerText">
<N8nText tag="h2" size="xlarge" bold>
{{ i18n.baseText('evaluation.collections.title') }}
</N8nText>
<N8nText size="xsmall" color="text-light">
{{ i18n.baseText('evaluation.collections.empty.subtitle') }}
<N8nText tag="p" size="small" color="text-base">
{{ i18n.baseText('evaluation.collections.subtitle') }}
</N8nText>
</div>
</div>
<N8nButton
variant="solid"
icon="plus"
:label="i18n.baseText('evaluation.collections.newCollection')"
data-test-id="eval-collections-new-button"
@click="onOpenWizard"
/>
</header>
<CollectionCard
v-for="collection in collections"
:key="collection.id"
:collection="collection"
:detail="store.getDetail(collection.id)"
<section :class="$style.section">
<div :class="$style.sectionHeader">
<N8nText tag="h3" size="medium" bold>
{{ i18n.baseText('evaluation.collections.section.collections') }}
</N8nText>
<N8nBadge theme="tertiary" size="small">{{ collections.length }}</N8nBadge>
</div>
<div v-if="collections.length === 0" :class="$style.emptyHint">
<N8nIcon icon="layers" size="medium" />
<div :class="$style.emptyHintBody">
<N8nText size="small" bold>
{{ i18n.baseText('evaluation.collections.empty.title') }}
</N8nText>
<N8nText size="xsmall" color="text-light">
{{ i18n.baseText('evaluation.collections.empty.subtitle') }}
</N8nText>
</div>
</div>
<CollectionCard
v-for="collection in pagedCollections"
:key="collection.id"
:collection="collection"
:detail="store.getDetail(collection.id)"
:workflow-id="workflowId"
:dataset-name="datasetNameByConfigId[collection.evaluationConfigId]"
/>
<N8nPagination
v-if="showCollectionsPagination"
:class="$style.pagination"
layout="prev, pager, next"
:current-page="collectionsPage"
:page-size="pageSize"
:total="collections.length"
@update:current-page="collectionsPage = $event"
/>
</section>
<section :class="$style.section">
<div :class="$style.sectionHeader">
<N8nText tag="h3" size="medium" bold>
{{ i18n.baseText('evaluation.collections.section.ungrouped') }}
</N8nText>
<N8nBadge theme="tertiary" size="small">{{ ungroupedRuns.length }}</N8nBadge>
<N8nText :class="$style.ungroupedHelper" size="xsmall" color="text-light">
{{ i18n.baseText('evaluation.collections.section.ungrouped.helper') }}
</N8nText>
</div>
<div v-if="ungroupedRuns.length === 0" :class="$style.emptyHint">
<N8nText size="small" color="text-light">
{{ i18n.baseText('evaluation.collections.section.ungrouped.empty') }}
</N8nText>
</div>
<UngroupedRunRow
v-for="run in ungroupedRuns"
:key="run.id"
:run="run"
:dataset-name-by-config-id="datasetNameByConfigId"
/>
</section>
<SetupCollectionWizard
:open="wizardOpen"
:workflow-id="workflowId"
:dataset-name="datasetNameByConfigId[collection.evaluationConfigId]"
@update:open="wizardOpen = $event"
@created="onCreated"
/>
</section>
<section :class="$style.section">
<div :class="$style.sectionHeader">
<N8nText tag="h3" size="medium" bold>
{{ i18n.baseText('evaluation.collections.section.ungrouped') }}
</N8nText>
<N8nBadge theme="tertiary" size="small">{{ ungroupedRuns.length }}</N8nBadge>
<N8nText :class="$style.ungroupedHelper" size="xsmall" color="text-light">
{{ i18n.baseText('evaluation.collections.section.ungrouped.helper') }}
</N8nText>
</div>
<div v-if="ungroupedRuns.length === 0" :class="$style.emptyHint">
<N8nText size="small" color="text-light">
{{ i18n.baseText('evaluation.collections.section.ungrouped.empty') }}
</N8nText>
</div>
<UngroupedRunRow
v-for="run in ungroupedRuns"
:key="run.id"
:run="run"
:dataset-name-by-config-id="datasetNameByConfigId"
/>
</section>
<SetupCollectionWizard
:open="wizardOpen"
:workflow-id="workflowId"
@update:open="wizardOpen = $event"
@created="onCreated"
/>
</div>
</div>
</template>
<style module lang="scss">
// Mirror EvaluationsView's `.wrapper` + `.content`: reserve the same 58px
// collapsed-sidebar gutter and centre the inner column, so this section's left
// and right edges line up with the runs section stacked above it.
.viewWrapper {
display: flex;
justify-content: center;
padding: 0 var(--spacing--lg);
padding-left: 58px;
}
.view {
width: 100%;
max-width: 1280px;
margin: 0 auto;
padding: var(--spacing--lg) var(--spacing--md);
// Match EvaluationsView's `.runs` cap so both stacked sections share width.
max-width: 1024px;
padding: var(--spacing--lg) 0;
display: flex;
flex-direction: column;
gap: var(--spacing--lg);
@@ -218,6 +269,12 @@ onBeforeUnmount(() => {
gap: var(--spacing--xs);
}
.pagination {
display: flex;
justify-content: center;
margin-top: var(--spacing--2xs);
}
.sectionHeader {
display: flex;
align-items: center;
@@ -1,8 +1,10 @@
<script setup lang="ts">
// Picks the list-view implementation based on the eval-collections rollout
// flag. Flag-off cohort sees the legacy `EvaluationsView` (single-run list);
// flag-on sees the collections list view. Sits at the `''` child route so
// `<RouterView>` in `EvaluationsRootView` resolves to whichever shape applies.
// The Evaluations tab stacks the config-eval hub (`EvaluationsView`: run test,
// config runs) above the eval-collections list when the rollout flag is on —
// config evals are the foundation, collections the comparison layer built on
// them. Both lists paginate to a few items so the two fit on one page without a
// long scroll. Flag-off shows just the config-eval hub, unpaginated. Sits at the
// `''` child route so `<RouterView>` in `EvaluationsRootView` resolves here.
import { defineAsyncComponent } from 'vue';
@@ -10,8 +12,7 @@ import { useEvalCollectionsFlag } from '../composables/useEvalCollectionsFlag';
import EvaluationsView from './EvaluationsView.vue';
// Lazy-load the collections surface so the flag-off cohort (the 0%-rollout
// default) never downloads the collections list, setup wizard, and chart
// graph. The legacy view stays eager — it's what most users land on.
// default) never downloads the collections list, setup wizard, and chart graph.
const EvalCollectionsListView = defineAsyncComponent(
async () => await import('./EvalCollectionsListView.vue'),
);
@@ -21,9 +22,25 @@ defineProps<{
}>();
const isCollectionsEnabled = useEvalCollectionsFlag();
// Per-list cap when both surfaces share the page, so neither pushes the other
// off-screen.
const STACKED_PAGE_SIZE = 5;
</script>
<template>
<EvalCollectionsListView v-if="isCollectionsEnabled" :workflow-id="workflowId" />
<EvaluationsView v-else :workflow-id="workflowId" />
<EvaluationsView v-if="!isCollectionsEnabled" :workflow-id="workflowId" />
<div v-else :class="$style.stack">
<EvaluationsView :workflow-id="workflowId" :runs-page-size="STACKED_PAGE_SIZE" />
<EvalCollectionsListView :workflow-id="workflowId" :page-size="STACKED_PAGE_SIZE" />
</div>
</template>
<style module lang="scss">
.stack {
display: flex;
flex-direction: column;
width: 100%;
}
</style>
@@ -17,6 +17,9 @@ import { N8nButton, N8nIcon, N8nPopover } from '@n8n/design-system';
const props = defineProps<{
workflowId: string;
// When set, the runs table paginates to this many rows per page (used when
// stacked alongside the collections list so both fit). Omitted → no paging.
runsPageSize?: number;
}>();
const locale = useI18n();
@@ -260,6 +263,7 @@ watch(runningTestRun, (run) => {
:class="$style.runs"
:runs="runs"
:workflow-id="props.workflowId"
:page-size="runsPageSize"
/>
</div>
</div>
@@ -0,0 +1,62 @@
import type { Locator } from '@playwright/test';
import { BasePage } from './BasePage';
/**
* The multi-version eval-collection compare view
* (`/workflow/:workflowId/evaluation/collections/:collectionId/compare`).
*/
export class EvaluationComparePage extends BasePage {
async goto(workflowId: string, collectionId: string): Promise<void> {
// The editor is an SPA — resolve on `domcontentloaded` rather than the full
// `load` event (which can lag on a cold route), then wait for the view.
await this.page.goto(`/workflow/${workflowId}/evaluation/collections/${collectionId}/compare`, {
waitUntil: 'domcontentloaded',
// Generous nav timeout: the dev server can cold-compile this route on
// first navigation. Harmless against the prebuilt editor in CI.
timeout: 60_000,
});
await this.getView().waitFor({ state: 'visible', timeout: 30_000 });
}
getView(): Locator {
return this.page.getByTestId('compare-collection-view');
}
getHeader(): Locator {
return this.page.getByTestId('compare-header');
}
getScoreChart(): Locator {
return this.page.getByTestId('compare-score-chart');
}
getTabs(): Locator {
return this.page.getByTestId('compare-tabs');
}
getDatasetMismatchBanner(): Locator {
return this.page.getByTestId('compare-dataset-mismatch');
}
getCasesTable(): Locator {
return this.page.getByTestId('compare-cases-table');
}
getCaseRows(): Locator {
return this.page.getByTestId('compare-cases-row');
}
getOutputsTab(): Locator {
return this.page.getByTestId('compare-outputs-tab');
}
getOutputColumns(): Locator {
return this.page.getByTestId('compare-outputs-column');
}
/** Click a case row to drill into its side-by-side outputs. */
async openCase(index: number): Promise<void> {
await this.getCaseRows().nth(index).click();
}
}
@@ -22,6 +22,7 @@ import { CredentialsPage } from './CredentialsPage';
import { DataTableDetails } from './DataTableDetails';
import { DataTableView } from './DataTableView';
import { DemoPage } from './DemoPage';
import { EvaluationComparePage } from './EvaluationComparePage';
import { ExecutionsPage } from './ExecutionsPage';
import { InstanceAiPage } from './InstanceAiPage';
import { KeycloakLoginPage } from './KeycloakLoginPage';
@@ -102,6 +103,7 @@ export class n8nPage {
readonly workflows: WorkflowsPage;
readonly notifications: NotificationsPage;
readonly credentials: CredentialsPage;
readonly evaluationCompare: EvaluationComparePage;
readonly executions: ExecutionsPage;
readonly sideBar: SidebarPage;
readonly dataTable: DataTableView;
@@ -185,6 +187,7 @@ export class n8nPage {
this.workflows = new WorkflowsPage(page);
this.notifications = new NotificationsPage(page);
this.credentials = new CredentialsPage(page);
this.evaluationCompare = new EvaluationComparePage(page);
this.executions = new ExecutionsPage(page);
this.sideBar = new SidebarPage(page);
this.signIn = new SignInPage(page);
@@ -0,0 +1,209 @@
import { nanoid } from 'nanoid';
import { test, expect } from '../../../fixtures/base';
import type { TestRequirements } from '../../../Types';
const COLLECTION_ID = 'col-e2e';
// Enable the eval-collections feature surface client-side; all eval REST calls
// are stubbed below, so no backend flag or real eval run is needed.
const requirements: TestRequirements = {
storage: {
N8N_EXPERIMENT_OVERRIDES: JSON.stringify({ '084_eval_collections': true }),
},
};
function caseFor(runId: string, runIndex: number, score: number, question: string, output: string) {
return {
id: `${runId}-${runIndex}`,
testRunId: runId,
executionId: null,
status: 'success',
createdAt: '2026-01-01T00:00:00Z',
updatedAt: '2026-01-01T00:00:00Z',
runAt: '2026-01-01T00:00:00Z',
runIndex,
metrics: { helpfulness: score },
inputs: { question },
outputs: { output },
};
}
// Per-run case executions. The "Capital of France?" case has the largest
// score spread (0.5 → 0.9), so the cases table — sorted by biggest regression
// first — puts it in the top row, which the drilldown assertion relies on.
const CASES: Record<string, Array<ReturnType<typeof caseFor>>> = {
'run-a': [
caseFor('run-a', 0, 0.5, 'Capital of France?', 'Paris'),
caseFor('run-a', 1, 0.8, 'What is 2+2?', '4'),
],
'run-b': [
caseFor('run-b', 0, 0.9, 'Capital of France?', 'The capital of France is Paris.'),
caseFor('run-b', 1, 0.88, 'What is 2+2?', '2 + 2 equals 4.'),
],
};
const json = (data: unknown) => ({
contentType: 'application/json',
body: JSON.stringify({ data }),
});
test.describe(
'Eval collection compare view @auth:owner',
{ annotation: [{ type: 'owner', description: 'AI' }] },
() => {
let workflowId: string;
test.beforeEach(async ({ n8n, setupRequirements }) => {
await setupRequirements(requirements);
const workflow = await n8n.api.workflows.createWorkflow({
name: `Compare E2E ${nanoid()}`,
nodes: [],
connections: {},
});
workflowId = workflow.id;
const record = {
id: COLLECTION_ID,
name: 'Tone tuning experiment',
description: null,
workflowId,
evaluationConfigId: 'cfg-1',
createdById: 'owner',
createdAt: '2026-01-01T00:00:00Z',
updatedAt: '2026-01-01T00:00:00Z',
runCount: 2,
};
const detail = {
...record,
runs: [
{
testRunId: 'run-a',
workflowVersionId: 'v1',
status: 'completed',
runAt: '2026-01-01T00:00:00Z',
completedAt: '2026-01-01T00:05:00Z',
avgScore: 0.61,
metrics: { helpfulness: 0.61 },
},
{
testRunId: 'run-b',
workflowVersionId: 'v2',
status: 'completed',
runAt: '2026-01-01T00:00:00Z',
completedAt: '2026-01-01T00:06:00Z',
avgScore: 0.89,
metrics: { helpfulness: 0.89 },
},
],
};
const insights = {
generatedAt: '2026-01-01T00:07:00Z',
modelUsed: 'test',
status: 'ok',
insights: {
winner: {
versionLabel: 'B',
headline: 'B wins',
body: 'Higher helpfulness across cases.',
},
regressions: [],
suggestedNext: {
headline: 'Try C',
body: 'Raise the temperature.',
hypothesis: 'More varied phrasing may help.',
},
},
};
// The evaluation root view only renders the compare route once the
// workflow has at least one test run (otherwise it shows the setup
// wizard), so stub the test-runs list too.
const testRuns = detail.runs.map((run) => ({
id: run.testRunId,
workflowId,
status: 'completed',
metrics: run.metrics,
createdAt: '2026-01-01T00:00:00Z',
updatedAt: '2026-01-01T00:06:00Z',
runAt: run.runAt,
completedAt: run.completedAt,
collectionId: COLLECTION_ID,
}));
await n8n.page.route(
new RegExp(`/rest/workflows/${workflowId}/test-runs(?!/)`),
async (route) => await route.fulfill(json(testRuns)),
);
// Order doesn't matter — the globs are disjoint (`(?!/)` keeps the detail
// and list routes from swallowing the /insights and /runs sub-paths).
await n8n.page.route(
new RegExp(`/rest/workflows/${workflowId}/eval-collections/${COLLECTION_ID}/insights`),
async (route) => await route.fulfill(json(insights)),
);
await n8n.page.route(
new RegExp(`/rest/workflows/${workflowId}/eval-collections/${COLLECTION_ID}(?!/)`),
async (route) => await route.fulfill(json(detail)),
);
await n8n.page.route(
new RegExp(`/rest/workflows/${workflowId}/eval-collections(?!/)`),
async (route) => await route.fulfill(json([record])),
);
await n8n.page.route(
new RegExp(`/rest/workflows/${workflowId}/test-runs/[^/]+/test-cases`),
async (route) => {
const runId =
route
.request()
.url()
.match(/test-runs\/([^/?]+)\/test-cases/)?.[1] ?? '';
await route.fulfill(json(CASES[runId] ?? []));
},
);
});
test('renders the hero and cases table, and drills into per-version outputs', async ({
n8n,
}) => {
const compare = n8n.evaluationCompare;
await compare.goto(workflowId, COLLECTION_ID);
await expect(compare.getHeader()).toContainText('Tone tuning experiment');
await expect(compare.getScoreChart()).toBeVisible();
await expect(compare.getTabs()).toBeVisible();
// Cases tab (default) lists both seeded cases by their input.
await expect(compare.getCasesTable()).toContainText('Capital of France?');
await expect(compare.getCasesTable()).toContainText('What is 2+2?');
// Drilling into a case row jumps to the side-by-side outputs, showing
// each version's distinct answer for that case.
await compare.openCase(0);
await expect(compare.getOutputsTab()).toBeVisible();
await expect(compare.getOutputsTab()).toContainText('Paris');
await expect(compare.getOutputsTab()).toContainText('The capital of France is Paris.');
});
test('surfaces a dataset-mismatch banner when run case counts diverge', async ({ n8n }) => {
// Re-stub run-b with a single case so the counts diverge (2 vs 1). A
// later route registration takes precedence in Playwright.
await n8n.page.route(
new RegExp(`/rest/workflows/${workflowId}/test-runs/[^/]+/test-cases`),
async (route) => {
const runId =
route
.request()
.url()
.match(/test-runs\/([^/?]+)\/test-cases/)?.[1] ?? '';
const cases = runId === 'run-b' ? CASES['run-b'].slice(0, 1) : (CASES[runId] ?? []);
await route.fulfill(json(cases));
},
);
const compare = n8n.evaluationCompare;
await compare.goto(workflowId, COLLECTION_ID);
await expect(compare.getDatasetMismatchBanner()).toBeVisible();
});
},
);