diff --git a/packages/@n8n/ai-workflow-builder.ee/evaluations/cli/runner.ts b/packages/@n8n/ai-workflow-builder.ee/evaluations/cli/runner.ts index 320d438fdce..26d8e28eabe 100644 --- a/packages/@n8n/ai-workflow-builder.ee/evaluations/cli/runner.ts +++ b/packages/@n8n/ai-workflow-builder.ee/evaluations/cli/runner.ts @@ -2,6 +2,7 @@ import pLimit from 'p-limit'; import pc from 'picocolors'; import { createProgressBar, updateProgress, displayResults, displayError } from './display.js'; +import type { BuilderFeatureFlags } from '../../src/workflow-builder-agent.js'; import { basicTestCases, generateTestCases } from '../chains/test-case-generator.js'; import { setupTestEnvironment, @@ -25,6 +26,7 @@ type CliEvaluationOptions = { testCaseFilter?: string; // Optional test case ID to run only a specific test testCases?: TestCase[]; // Optional array of test cases to run (if not provided, uses defaults and generation) repetitions?: number; // Number of times to run each test (e.g. for cache warming analysis) + featureFlags?: BuilderFeatureFlags; // Optional feature flags to pass to the agent (e.g. templateExamples, multiAgent) }; /** @@ -32,12 +34,20 @@ type CliEvaluationOptions = { * Supports concurrency control via EVALUATION_CONCURRENCY environment variable */ export async function runCliEvaluation(options: CliEvaluationOptions = {}): Promise { - const { repetitions = 1, testCaseFilter } = options; + const { repetitions = 1, testCaseFilter, featureFlags } = options; console.log(formatHeader('AI Workflow Builder Full Evaluation', 70)); if (repetitions > 1) { console.log(pc.yellow(`➔ Each test will be run ${repetitions} times for cache analysis`)); } + if (featureFlags) { + const enabledFlags = Object.entries(featureFlags) + .filter(([, v]) => v === true) + .map(([k]) => k); + if (enabledFlags.length > 0) { + console.log(pc.green(`➔ Feature flags enabled: ${enabledFlags.join(', ')}`)); + } + } console.log(); try { // Setup test environment @@ -105,7 +115,9 @@ export async function runCliEvaluation(options: CliEvaluationOptions = {}): Prom // Create a dedicated agent for this test to avoid state conflicts const testAgent = createAgent(parsedNodeTypes, llm, tracer); - const result = await runSingleTest(testAgent, llm, testCase, parsedNodeTypes); + const result = await runSingleTest(testAgent, llm, testCase, parsedNodeTypes, { + featureFlags, + }); testResults[testCase.id] = result.error ? 'fail' : 'pass'; completed++; diff --git a/packages/@n8n/ai-workflow-builder.ee/evaluations/core/test-runner.ts b/packages/@n8n/ai-workflow-builder.ee/evaluations/core/test-runner.ts index 3cdfbe0d488..aa814eedf82 100644 --- a/packages/@n8n/ai-workflow-builder.ee/evaluations/core/test-runner.ts +++ b/packages/@n8n/ai-workflow-builder.ee/evaluations/core/test-runner.ts @@ -1,7 +1,7 @@ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'; import type { INodeTypeDescription } from 'n8n-workflow'; -import type { WorkflowBuilderAgent } from '../../src/workflow-builder-agent'; +import type { BuilderFeatureFlags, WorkflowBuilderAgent } from '../../src/workflow-builder-agent'; import { evaluateWorkflow } from '../chains/workflow-evaluator'; import { programmaticEvaluation } from '../programmatic/programmatic-evaluation'; import type { EvaluationInput, TestCase } from '../types/evaluation'; @@ -69,12 +69,22 @@ export function createErrorResult(testCase: TestCase, error: unknown): TestResul }; } +export interface RunSingleTestOptions { + agent: WorkflowBuilderAgent; + llm: BaseChatModel; + testCase: TestCase; + nodeTypes: INodeTypeDescription[]; + userId?: string; + featureFlags?: BuilderFeatureFlags; +} + /** * Runs a single test case by generating a workflow and evaluating it * @param agent - The workflow builder agent to use * @param llm - Language model for evaluation * @param testCase - Test case to execute - * @param userId - User ID for the session + * @param nodeTypes - Array of node type descriptions + * @params opts - userId, User ID for the session and featureFlags, Optional feature flags to pass to the agent * @returns Test result with generated workflow and evaluation */ export async function runSingleTest( @@ -82,12 +92,15 @@ export async function runSingleTest( llm: BaseChatModel, testCase: TestCase, nodeTypes: INodeTypeDescription[], - userId: string = 'test-user', + opts?: { userId?: string; featureFlags?: BuilderFeatureFlags }, ): Promise { + const userId = opts?.userId ?? 'test-user'; try { // Generate workflow const startTime = Date.now(); - await consumeGenerator(agent.chat(getChatPayload(testCase.prompt, testCase.id), userId)); + await consumeGenerator( + agent.chat(getChatPayload(testCase.prompt, testCase.id, opts?.featureFlags), userId), + ); const generationTime = Date.now() - startTime; // Get generated workflow with validation diff --git a/packages/@n8n/ai-workflow-builder.ee/evaluations/index.ts b/packages/@n8n/ai-workflow-builder.ee/evaluations/index.ts index 10c6aac5f9c..162246f44d6 100644 --- a/packages/@n8n/ai-workflow-builder.ee/evaluations/index.ts +++ b/packages/@n8n/ai-workflow-builder.ee/evaluations/index.ts @@ -1,3 +1,5 @@ +import type { BuilderFeatureFlags } from '@/workflow-builder-agent'; + import { runCliEvaluation } from './cli/runner.js'; import { runPairwiseLangsmithEvaluation } from './langsmith/pairwise-runner.js'; import { runLangsmithEvaluation } from './langsmith/runner.js'; @@ -36,13 +38,21 @@ async function main(): Promise { : 1; const repetitions = Number.isNaN(repetitionsArg) ? 1 : repetitionsArg; + // Parse feature flags from environment variables or CLI arguments + const featureFlags = parseFeatureFlags(); + if (usePairwiseEval) { - await runPairwiseLangsmithEvaluation(repetitions); + await runPairwiseLangsmithEvaluation(repetitions, featureFlags); } else if (useLangsmith) { - await runLangsmithEvaluation(repetitions); + await runLangsmithEvaluation(repetitions, featureFlags); } else { const csvTestCases = promptsCsvPath ? loadTestCasesFromCsv(promptsCsvPath) : undefined; - await runCliEvaluation({ testCases: csvTestCases, testCaseFilter: testCaseId, repetitions }); + await runCliEvaluation({ + testCases: csvTestCases, + testCaseFilter: testCaseId, + repetitions, + featureFlags, + }); } } @@ -68,6 +78,36 @@ function getFlagValue(flag: string): string | undefined { return undefined; } +/** + * Parse feature flags from environment variables or CLI arguments. + * Environment variables: + * - EVAL_FEATURE_TEMPLATE_EXAMPLES=true - Enable template examples feature + * - EVAL_FEATURE_MULTI_AGENT=true - Enable multi-agent feature + * CLI arguments: + * - --template-examples - Enable template examples feature + * - --multi-agent - Enable multi-agent feature + */ +function parseFeatureFlags(): BuilderFeatureFlags | undefined { + const templateExamplesFromEnv = process.env.EVAL_FEATURE_TEMPLATE_EXAMPLES === 'true'; + const multiAgentFromEnv = process.env.EVAL_FEATURE_MULTI_AGENT === 'true'; + + const templateExamplesFromCli = process.argv.includes('--template-examples'); + const multiAgentFromCli = process.argv.includes('--multi-agent'); + + const templateExamples = templateExamplesFromEnv || templateExamplesFromCli; + const multiAgent = multiAgentFromEnv || multiAgentFromCli; + + // Only return feature flags object if at least one flag is set + if (templateExamples || multiAgent) { + return { + templateExamples: templateExamples || undefined, + multiAgent: multiAgent || undefined, + }; + } + + return undefined; +} + // Run if called directly if (require.main === module) { main().catch(console.error); diff --git a/packages/@n8n/ai-workflow-builder.ee/evaluations/langsmith/pairwise-runner.ts b/packages/@n8n/ai-workflow-builder.ee/evaluations/langsmith/pairwise-runner.ts index 7146569acb2..89187b1138d 100644 --- a/packages/@n8n/ai-workflow-builder.ee/evaluations/langsmith/pairwise-runner.ts +++ b/packages/@n8n/ai-workflow-builder.ee/evaluations/langsmith/pairwise-runner.ts @@ -7,6 +7,7 @@ import type { INodeTypeDescription } from 'n8n-workflow'; import pc from 'picocolors'; import type { SimpleWorkflow } from '../../src/types/workflow'; +import type { BuilderFeatureFlags } from '../../src/workflow-builder-agent'; import { evaluateWorkflowPairwise } from '../chains/pairwise-evaluator'; import { setupTestEnvironment, createAgent } from '../core/environment'; import { generateRunId, isWorkflowStateValues } from '../types/langsmith'; @@ -41,6 +42,7 @@ function createPairwiseWorkflowGenerator( parsedNodeTypes: INodeTypeDescription[], llm: BaseChatModel, tracer?: LangChainTracer, + featureFlags?: BuilderFeatureFlags, ) { return async (inputs: PairwiseDatasetInput) => { const runId = generateRunId(); @@ -50,7 +52,10 @@ function createPairwiseWorkflowGenerator( // Use the prompt from the dataset await consumeGenerator( - agent.chat(getChatPayload(inputs.prompt, runId), 'langsmith-pairwise-eval-user'), + agent.chat( + getChatPayload(inputs.prompt, runId, featureFlags), + 'langsmith-pairwise-eval-user', + ), ); // Get generated workflow @@ -117,9 +122,21 @@ function createPairwiseLangsmithEvaluator(llm: BaseChatModel) { }; } -export async function runPairwiseLangsmithEvaluation(repetitions: number = 1): Promise { +export async function runPairwiseLangsmithEvaluation( + repetitions: number = 1, + featureFlags?: BuilderFeatureFlags, +): Promise { console.log(formatHeader('AI Workflow Builder Pairwise Evaluation', 70)); + if (featureFlags) { + const enabledFlags = Object.entries(featureFlags) + .filter(([, v]) => v === true) + .map(([k]) => k); + if (enabledFlags.length > 0) { + console.log(pc.green(`➔ Feature flags enabled: ${enabledFlags.join(', ')}`)); + } + } + if (!process.env.LANGSMITH_API_KEY) { console.error(pc.red('✗ LANGSMITH_API_KEY environment variable not set')); process.exit(1); @@ -161,7 +178,12 @@ export async function runPairwiseLangsmithEvaluation(repetitions: number = 1): P data = examples; } - const generateWorkflow = createPairwiseWorkflowGenerator(parsedNodeTypes, llm, tracer); + const generateWorkflow = createPairwiseWorkflowGenerator( + parsedNodeTypes, + llm, + tracer, + featureFlags, + ); const evaluator = createPairwiseLangsmithEvaluator(llm); await evaluate(generateWorkflow, { diff --git a/packages/@n8n/ai-workflow-builder.ee/evaluations/langsmith/runner.ts b/packages/@n8n/ai-workflow-builder.ee/evaluations/langsmith/runner.ts index 8ea144ab2b9..f0c6d91e35c 100644 --- a/packages/@n8n/ai-workflow-builder.ee/evaluations/langsmith/runner.ts +++ b/packages/@n8n/ai-workflow-builder.ee/evaluations/langsmith/runner.ts @@ -5,6 +5,7 @@ import type { INodeTypeDescription } from 'n8n-workflow'; import pc from 'picocolors'; import { createLangsmithEvaluator } from './evaluator'; +import type { BuilderFeatureFlags } from '../../src/workflow-builder-agent'; import type { WorkflowState } from '../../src/workflow-state'; import { setupTestEnvironment, createAgent } from '../core/environment'; import { @@ -20,12 +21,14 @@ import { consumeGenerator, formatHeader, getChatPayload } from '../utils/evaluat * @param parsedNodeTypes - Node types * @param llm - Language model * @param tracer - Optional tracer + * @param featureFlags - Optional feature flags to pass to the agent * @returns Function that generates workflows from inputs */ function createWorkflowGenerator( parsedNodeTypes: INodeTypeDescription[], llm: BaseChatModel, tracer?: LangChainTracer, + featureFlags?: BuilderFeatureFlags, ) { return async (inputs: typeof WorkflowState.State) => { // Generate a unique ID for this evaluation run @@ -43,7 +46,7 @@ function createWorkflowGenerator( // Create agent for this run const agent = createAgent(parsedNodeTypes, llm, tracer); await consumeGenerator( - agent.chat(getChatPayload(messageContent, runId), 'langsmith-eval-user'), + agent.chat(getChatPayload(messageContent, runId, featureFlags), 'langsmith-eval-user'), ); // Get generated workflow with validation @@ -75,12 +78,24 @@ function createWorkflowGenerator( /** * Runs evaluation using Langsmith * @param repetitions - Number of times to run each example (default: 1) + * @param featureFlags - Optional feature flags to pass to the agent */ -export async function runLangsmithEvaluation(repetitions: number = 1): Promise { +export async function runLangsmithEvaluation( + repetitions: number = 1, + featureFlags?: BuilderFeatureFlags, +): Promise { console.log(formatHeader('AI Workflow Builder Langsmith Evaluation', 70)); if (repetitions > 1) { console.log(pc.yellow(`➔ Each example will be run ${repetitions} times`)); } + if (featureFlags) { + const enabledFlags = Object.entries(featureFlags) + .filter(([, v]) => v === true) + .map(([k]) => k); + if (enabledFlags.length > 0) { + console.log(pc.green(`➔ Feature flags enabled: ${enabledFlags.join(', ')}`)); + } + } console.log(); // Check for Langsmith API key @@ -123,7 +138,7 @@ export async function runLangsmithEvaluation(repetitions: number = 1): Promise(gen: AsyncGenerator) { } } -export function getChatPayload(message: string, id: string): ChatPayload { +export function getChatPayload( + message: string, + id: string, + featureFlags?: BuilderFeatureFlags, +): ChatPayload { return { message, workflowContext: { currentWorkflow: { id, nodes: [], connections: {} }, }, + featureFlags, }; }