mirror of
https://github.com/n8n-io/n8n.git
synced 2026-09-24 23:22:38 +08:00
fix(ai-builder): Allow setting evaluation feature flags via environment variables (no-changelog) (#22813)
This commit is contained in:
@@ -2,6 +2,7 @@ import pLimit from 'p-limit';
|
||||
import pc from 'picocolors';
|
||||
|
||||
import { createProgressBar, updateProgress, displayResults, displayError } from './display.js';
|
||||
import type { BuilderFeatureFlags } from '../../src/workflow-builder-agent.js';
|
||||
import { basicTestCases, generateTestCases } from '../chains/test-case-generator.js';
|
||||
import {
|
||||
setupTestEnvironment,
|
||||
@@ -25,6 +26,7 @@ type CliEvaluationOptions = {
|
||||
testCaseFilter?: string; // Optional test case ID to run only a specific test
|
||||
testCases?: TestCase[]; // Optional array of test cases to run (if not provided, uses defaults and generation)
|
||||
repetitions?: number; // Number of times to run each test (e.g. for cache warming analysis)
|
||||
featureFlags?: BuilderFeatureFlags; // Optional feature flags to pass to the agent (e.g. templateExamples, multiAgent)
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -32,12 +34,20 @@ type CliEvaluationOptions = {
|
||||
* Supports concurrency control via EVALUATION_CONCURRENCY environment variable
|
||||
*/
|
||||
export async function runCliEvaluation(options: CliEvaluationOptions = {}): Promise<void> {
|
||||
const { repetitions = 1, testCaseFilter } = options;
|
||||
const { repetitions = 1, testCaseFilter, featureFlags } = options;
|
||||
|
||||
console.log(formatHeader('AI Workflow Builder Full Evaluation', 70));
|
||||
if (repetitions > 1) {
|
||||
console.log(pc.yellow(`➔ Each test will be run ${repetitions} times for cache analysis`));
|
||||
}
|
||||
if (featureFlags) {
|
||||
const enabledFlags = Object.entries(featureFlags)
|
||||
.filter(([, v]) => v === true)
|
||||
.map(([k]) => k);
|
||||
if (enabledFlags.length > 0) {
|
||||
console.log(pc.green(`➔ Feature flags enabled: ${enabledFlags.join(', ')}`));
|
||||
}
|
||||
}
|
||||
console.log();
|
||||
try {
|
||||
// Setup test environment
|
||||
@@ -105,7 +115,9 @@ export async function runCliEvaluation(options: CliEvaluationOptions = {}): Prom
|
||||
|
||||
// Create a dedicated agent for this test to avoid state conflicts
|
||||
const testAgent = createAgent(parsedNodeTypes, llm, tracer);
|
||||
const result = await runSingleTest(testAgent, llm, testCase, parsedNodeTypes);
|
||||
const result = await runSingleTest(testAgent, llm, testCase, parsedNodeTypes, {
|
||||
featureFlags,
|
||||
});
|
||||
|
||||
testResults[testCase.id] = result.error ? 'fail' : 'pass';
|
||||
completed++;
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
|
||||
import type { INodeTypeDescription } from 'n8n-workflow';
|
||||
|
||||
import type { WorkflowBuilderAgent } from '../../src/workflow-builder-agent';
|
||||
import type { BuilderFeatureFlags, WorkflowBuilderAgent } from '../../src/workflow-builder-agent';
|
||||
import { evaluateWorkflow } from '../chains/workflow-evaluator';
|
||||
import { programmaticEvaluation } from '../programmatic/programmatic-evaluation';
|
||||
import type { EvaluationInput, TestCase } from '../types/evaluation';
|
||||
@@ -69,12 +69,22 @@ export function createErrorResult(testCase: TestCase, error: unknown): TestResul
|
||||
};
|
||||
}
|
||||
|
||||
export interface RunSingleTestOptions {
|
||||
agent: WorkflowBuilderAgent;
|
||||
llm: BaseChatModel;
|
||||
testCase: TestCase;
|
||||
nodeTypes: INodeTypeDescription[];
|
||||
userId?: string;
|
||||
featureFlags?: BuilderFeatureFlags;
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs a single test case by generating a workflow and evaluating it
|
||||
* @param agent - The workflow builder agent to use
|
||||
* @param llm - Language model for evaluation
|
||||
* @param testCase - Test case to execute
|
||||
* @param userId - User ID for the session
|
||||
* @param nodeTypes - Array of node type descriptions
|
||||
* @params opts - userId, User ID for the session and featureFlags, Optional feature flags to pass to the agent
|
||||
* @returns Test result with generated workflow and evaluation
|
||||
*/
|
||||
export async function runSingleTest(
|
||||
@@ -82,12 +92,15 @@ export async function runSingleTest(
|
||||
llm: BaseChatModel,
|
||||
testCase: TestCase,
|
||||
nodeTypes: INodeTypeDescription[],
|
||||
userId: string = 'test-user',
|
||||
opts?: { userId?: string; featureFlags?: BuilderFeatureFlags },
|
||||
): Promise<TestResult> {
|
||||
const userId = opts?.userId ?? 'test-user';
|
||||
try {
|
||||
// Generate workflow
|
||||
const startTime = Date.now();
|
||||
await consumeGenerator(agent.chat(getChatPayload(testCase.prompt, testCase.id), userId));
|
||||
await consumeGenerator(
|
||||
agent.chat(getChatPayload(testCase.prompt, testCase.id, opts?.featureFlags), userId),
|
||||
);
|
||||
const generationTime = Date.now() - startTime;
|
||||
|
||||
// Get generated workflow with validation
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
import type { BuilderFeatureFlags } from '@/workflow-builder-agent';
|
||||
|
||||
import { runCliEvaluation } from './cli/runner.js';
|
||||
import { runPairwiseLangsmithEvaluation } from './langsmith/pairwise-runner.js';
|
||||
import { runLangsmithEvaluation } from './langsmith/runner.js';
|
||||
@@ -36,13 +38,21 @@ async function main(): Promise<void> {
|
||||
: 1;
|
||||
const repetitions = Number.isNaN(repetitionsArg) ? 1 : repetitionsArg;
|
||||
|
||||
// Parse feature flags from environment variables or CLI arguments
|
||||
const featureFlags = parseFeatureFlags();
|
||||
|
||||
if (usePairwiseEval) {
|
||||
await runPairwiseLangsmithEvaluation(repetitions);
|
||||
await runPairwiseLangsmithEvaluation(repetitions, featureFlags);
|
||||
} else if (useLangsmith) {
|
||||
await runLangsmithEvaluation(repetitions);
|
||||
await runLangsmithEvaluation(repetitions, featureFlags);
|
||||
} else {
|
||||
const csvTestCases = promptsCsvPath ? loadTestCasesFromCsv(promptsCsvPath) : undefined;
|
||||
await runCliEvaluation({ testCases: csvTestCases, testCaseFilter: testCaseId, repetitions });
|
||||
await runCliEvaluation({
|
||||
testCases: csvTestCases,
|
||||
testCaseFilter: testCaseId,
|
||||
repetitions,
|
||||
featureFlags,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -68,6 +78,36 @@ function getFlagValue(flag: string): string | undefined {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse feature flags from environment variables or CLI arguments.
|
||||
* Environment variables:
|
||||
* - EVAL_FEATURE_TEMPLATE_EXAMPLES=true - Enable template examples feature
|
||||
* - EVAL_FEATURE_MULTI_AGENT=true - Enable multi-agent feature
|
||||
* CLI arguments:
|
||||
* - --template-examples - Enable template examples feature
|
||||
* - --multi-agent - Enable multi-agent feature
|
||||
*/
|
||||
function parseFeatureFlags(): BuilderFeatureFlags | undefined {
|
||||
const templateExamplesFromEnv = process.env.EVAL_FEATURE_TEMPLATE_EXAMPLES === 'true';
|
||||
const multiAgentFromEnv = process.env.EVAL_FEATURE_MULTI_AGENT === 'true';
|
||||
|
||||
const templateExamplesFromCli = process.argv.includes('--template-examples');
|
||||
const multiAgentFromCli = process.argv.includes('--multi-agent');
|
||||
|
||||
const templateExamples = templateExamplesFromEnv || templateExamplesFromCli;
|
||||
const multiAgent = multiAgentFromEnv || multiAgentFromCli;
|
||||
|
||||
// Only return feature flags object if at least one flag is set
|
||||
if (templateExamples || multiAgent) {
|
||||
return {
|
||||
templateExamples: templateExamples || undefined,
|
||||
multiAgent: multiAgent || undefined,
|
||||
};
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
// Run if called directly
|
||||
if (require.main === module) {
|
||||
main().catch(console.error);
|
||||
|
||||
@@ -7,6 +7,7 @@ import type { INodeTypeDescription } from 'n8n-workflow';
|
||||
import pc from 'picocolors';
|
||||
|
||||
import type { SimpleWorkflow } from '../../src/types/workflow';
|
||||
import type { BuilderFeatureFlags } from '../../src/workflow-builder-agent';
|
||||
import { evaluateWorkflowPairwise } from '../chains/pairwise-evaluator';
|
||||
import { setupTestEnvironment, createAgent } from '../core/environment';
|
||||
import { generateRunId, isWorkflowStateValues } from '../types/langsmith';
|
||||
@@ -41,6 +42,7 @@ function createPairwiseWorkflowGenerator(
|
||||
parsedNodeTypes: INodeTypeDescription[],
|
||||
llm: BaseChatModel,
|
||||
tracer?: LangChainTracer,
|
||||
featureFlags?: BuilderFeatureFlags,
|
||||
) {
|
||||
return async (inputs: PairwiseDatasetInput) => {
|
||||
const runId = generateRunId();
|
||||
@@ -50,7 +52,10 @@ function createPairwiseWorkflowGenerator(
|
||||
|
||||
// Use the prompt from the dataset
|
||||
await consumeGenerator(
|
||||
agent.chat(getChatPayload(inputs.prompt, runId), 'langsmith-pairwise-eval-user'),
|
||||
agent.chat(
|
||||
getChatPayload(inputs.prompt, runId, featureFlags),
|
||||
'langsmith-pairwise-eval-user',
|
||||
),
|
||||
);
|
||||
|
||||
// Get generated workflow
|
||||
@@ -117,9 +122,21 @@ function createPairwiseLangsmithEvaluator(llm: BaseChatModel) {
|
||||
};
|
||||
}
|
||||
|
||||
export async function runPairwiseLangsmithEvaluation(repetitions: number = 1): Promise<void> {
|
||||
export async function runPairwiseLangsmithEvaluation(
|
||||
repetitions: number = 1,
|
||||
featureFlags?: BuilderFeatureFlags,
|
||||
): Promise<void> {
|
||||
console.log(formatHeader('AI Workflow Builder Pairwise Evaluation', 70));
|
||||
|
||||
if (featureFlags) {
|
||||
const enabledFlags = Object.entries(featureFlags)
|
||||
.filter(([, v]) => v === true)
|
||||
.map(([k]) => k);
|
||||
if (enabledFlags.length > 0) {
|
||||
console.log(pc.green(`➔ Feature flags enabled: ${enabledFlags.join(', ')}`));
|
||||
}
|
||||
}
|
||||
|
||||
if (!process.env.LANGSMITH_API_KEY) {
|
||||
console.error(pc.red('✗ LANGSMITH_API_KEY environment variable not set'));
|
||||
process.exit(1);
|
||||
@@ -161,7 +178,12 @@ export async function runPairwiseLangsmithEvaluation(repetitions: number = 1): P
|
||||
data = examples;
|
||||
}
|
||||
|
||||
const generateWorkflow = createPairwiseWorkflowGenerator(parsedNodeTypes, llm, tracer);
|
||||
const generateWorkflow = createPairwiseWorkflowGenerator(
|
||||
parsedNodeTypes,
|
||||
llm,
|
||||
tracer,
|
||||
featureFlags,
|
||||
);
|
||||
const evaluator = createPairwiseLangsmithEvaluator(llm);
|
||||
|
||||
await evaluate(generateWorkflow, {
|
||||
|
||||
@@ -5,6 +5,7 @@ import type { INodeTypeDescription } from 'n8n-workflow';
|
||||
import pc from 'picocolors';
|
||||
|
||||
import { createLangsmithEvaluator } from './evaluator';
|
||||
import type { BuilderFeatureFlags } from '../../src/workflow-builder-agent';
|
||||
import type { WorkflowState } from '../../src/workflow-state';
|
||||
import { setupTestEnvironment, createAgent } from '../core/environment';
|
||||
import {
|
||||
@@ -20,12 +21,14 @@ import { consumeGenerator, formatHeader, getChatPayload } from '../utils/evaluat
|
||||
* @param parsedNodeTypes - Node types
|
||||
* @param llm - Language model
|
||||
* @param tracer - Optional tracer
|
||||
* @param featureFlags - Optional feature flags to pass to the agent
|
||||
* @returns Function that generates workflows from inputs
|
||||
*/
|
||||
function createWorkflowGenerator(
|
||||
parsedNodeTypes: INodeTypeDescription[],
|
||||
llm: BaseChatModel,
|
||||
tracer?: LangChainTracer,
|
||||
featureFlags?: BuilderFeatureFlags,
|
||||
) {
|
||||
return async (inputs: typeof WorkflowState.State) => {
|
||||
// Generate a unique ID for this evaluation run
|
||||
@@ -43,7 +46,7 @@ function createWorkflowGenerator(
|
||||
// Create agent for this run
|
||||
const agent = createAgent(parsedNodeTypes, llm, tracer);
|
||||
await consumeGenerator(
|
||||
agent.chat(getChatPayload(messageContent, runId), 'langsmith-eval-user'),
|
||||
agent.chat(getChatPayload(messageContent, runId, featureFlags), 'langsmith-eval-user'),
|
||||
);
|
||||
|
||||
// Get generated workflow with validation
|
||||
@@ -75,12 +78,24 @@ function createWorkflowGenerator(
|
||||
/**
|
||||
* Runs evaluation using Langsmith
|
||||
* @param repetitions - Number of times to run each example (default: 1)
|
||||
* @param featureFlags - Optional feature flags to pass to the agent
|
||||
*/
|
||||
export async function runLangsmithEvaluation(repetitions: number = 1): Promise<void> {
|
||||
export async function runLangsmithEvaluation(
|
||||
repetitions: number = 1,
|
||||
featureFlags?: BuilderFeatureFlags,
|
||||
): Promise<void> {
|
||||
console.log(formatHeader('AI Workflow Builder Langsmith Evaluation', 70));
|
||||
if (repetitions > 1) {
|
||||
console.log(pc.yellow(`➔ Each example will be run ${repetitions} times`));
|
||||
}
|
||||
if (featureFlags) {
|
||||
const enabledFlags = Object.entries(featureFlags)
|
||||
.filter(([, v]) => v === true)
|
||||
.map(([k]) => k);
|
||||
if (enabledFlags.length > 0) {
|
||||
console.log(pc.green(`➔ Feature flags enabled: ${enabledFlags.join(', ')}`));
|
||||
}
|
||||
}
|
||||
console.log();
|
||||
|
||||
// Check for Langsmith API key
|
||||
@@ -123,7 +138,7 @@ export async function runLangsmithEvaluation(repetitions: number = 1): Promise<v
|
||||
const startTime = Date.now();
|
||||
|
||||
// Create workflow generation function
|
||||
const generateWorkflow = createWorkflowGenerator(parsedNodeTypes, llm, tracer);
|
||||
const generateWorkflow = createWorkflowGenerator(parsedNodeTypes, llm, tracer, featureFlags);
|
||||
|
||||
// Create evaluator with both LLM-based and programmatic evaluation
|
||||
const evaluator = createLangsmithEvaluator(llm, parsedNodeTypes);
|
||||
|
||||
@@ -8,7 +8,7 @@ import { join } from 'path';
|
||||
import pc from 'picocolors';
|
||||
|
||||
import { anthropicClaudeSonnet45 } from '../../src/llm-config';
|
||||
import type { ChatPayload } from '../../src/workflow-builder-agent';
|
||||
import type { BuilderFeatureFlags, ChatPayload } from '../../src/workflow-builder-agent';
|
||||
import { WorkflowBuilderAgent } from '../../src/workflow-builder-agent';
|
||||
import type { Violation } from '../types/evaluation';
|
||||
import type { TestResult } from '../types/test-result';
|
||||
@@ -277,11 +277,16 @@ export async function consumeGenerator<T>(gen: AsyncGenerator<T>) {
|
||||
}
|
||||
}
|
||||
|
||||
export function getChatPayload(message: string, id: string): ChatPayload {
|
||||
export function getChatPayload(
|
||||
message: string,
|
||||
id: string,
|
||||
featureFlags?: BuilderFeatureFlags,
|
||||
): ChatPayload {
|
||||
return {
|
||||
message,
|
||||
workflowContext: {
|
||||
currentWorkflow: { id, nodes: [], connections: {} },
|
||||
},
|
||||
featureFlags,
|
||||
};
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user