From aa26ee2eeb01a71737e1e0fe32241f5c4df337cb Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Mon, 10 Mar 2025 18:03:09 -0700 Subject: [PATCH] fix: update evaluator to align with new responseFormat logic --- blocks/blocks/evaluator.ts | 123 ++++++++++++++++++++++---- executor/handlers.ts | 176 +++++++++++++++++++++++++++++++++---- providers/index.ts | 7 +- providers/utils.ts | 7 ++ tools/index.ts | 1 - 5 files changed, 279 insertions(+), 35 deletions(-) diff --git a/blocks/blocks/evaluator.ts b/blocks/blocks/evaluator.ts index b2640fba61..b8d644d6ee 100644 --- a/blocks/blocks/evaluator.ts +++ b/blocks/blocks/evaluator.ts @@ -27,34 +27,93 @@ interface EvaluatorResponse extends ToolResponse { } export const generateEvaluatorPrompt = (metrics: Metric[], content: string): string => { + // Create a clear metrics description with name, range, and description const metricsDescription = metrics .map( - (metric) => `${metric.name} (${metric.range.min}-${metric.range.max}): ${metric.description}` + (metric) => + `"${metric.name}" (${metric.range.min}-${metric.range.max}): ${metric.description}` ) .join('\n') + // Format the content properly - try to detect and format JSON + let formattedContent = content + try { + // If content looks like JSON (starts with { or [) + if ( + typeof content === 'string' && + (content.trim().startsWith('{') || content.trim().startsWith('[')) + ) { + // Try to parse and pretty-print + const parsedContent = JSON.parse(content) + formattedContent = JSON.stringify(parsedContent, null, 2) + } + // If it's already an object (shouldn't happen here but just in case) + else if (typeof content === 'object') { + formattedContent = JSON.stringify(content, null, 2) + } + } catch (e) { + console.warn('Warning: Content may not be valid JSON, using as-is', e) + formattedContent = content + } + + // Generate an example of the expected output format + const exampleOutput = metrics.reduce( + (acc, metric) => { + acc[metric.name] = Math.floor((metric.range.min + metric.range.max) / 2) // Use middle of range as example + return acc + }, + {} as Record + ) + return `You are an objective evaluation agent. Analyze the content against the provided metrics and provide detailed scoring. Evaluation Instructions: +- You MUST evaluate the content against each metric - For each metric, provide a numeric score within the specified range -- Your response must be a valid JSON object with each metric as a number field +- Your response MUST be a valid JSON object with each metric name as a key and a numeric score as the value +- Use EXACTLY the metric names provided (case-sensitive, no modifications) +- Follow the exact schema of the response format provided to you - Do not include explanations in the JSON - only numeric scores -- Under any circumstances, do not include any text before or after the JSON object +- Do not add any additional fields not specified in the schema +- Do not include ANY text before or after the JSON object + Metrics to evaluate: ${metricsDescription} Content to evaluate: -${content}` +${formattedContent} + +Example of expected response format (with different scores): +${JSON.stringify(exampleOutput, null, 2)} + +Remember: Your response MUST be a valid JSON object containing only the metrics as keys with their numeric scores as values. No text explanations.` } // Simplified response format generator that matches the agent block schema structure -const generateResponseFormat = (metrics: Metric[]) => ({ - fields: metrics.map((metric) => ({ - name: metric.name, - type: 'number', - description: `${metric.description} (Score between ${metric.range.min}-${metric.range.max})`, - })), -}) +const generateResponseFormat = (metrics: Metric[]) => { + // Create properties for each metric + const properties: Record = {} + + // Add each metric as a property + metrics.forEach((metric) => { + properties[metric.name] = { + type: 'number', + description: `${metric.description} (Score between ${metric.range.min}-${metric.range.max})`, + } + }) + + // Return a proper JSON Schema format + return { + name: 'evaluation_response', + schema: { + type: 'object', + properties, + required: metrics.map((metric) => metric.name), + additionalProperties: false, + }, + strict: true, + } +} export const EvaluatorBlock: BlockConfig = { type: 'evaluator', @@ -102,14 +161,42 @@ export const EvaluatorBlock: BlockConfig = { layout: 'full', hidden: true, value: (params: Record) => { - const metrics = params.metrics || [] - const content = params.content || '' - const responseFormat = generateResponseFormat(metrics) + try { + const metrics = params.metrics || [] - return JSON.stringify({ - systemPrompt: generateEvaluatorPrompt(metrics, content), - responseFormat, - }) + // Process content safely + let processedContent = '' + if (typeof params.content === 'object') { + processedContent = JSON.stringify(params.content, null, 2) + } else { + processedContent = String(params.content || '') + } + + // Generate prompt and response format directly + const promptText = generateEvaluatorPrompt(metrics, processedContent) + const responseFormatObj = generateResponseFormat(metrics) + + // Create a clean, simple JSON object + const result = { + systemPrompt: promptText, + responseFormat: responseFormatObj, + } + + return JSON.stringify(result) + } catch (e) { + console.error('Error in systemPrompt value function:', e) + // Return a minimal valid JSON as fallback + return JSON.stringify({ + systemPrompt: 'Evaluate the content and return a JSON with metric scores.', + responseFormat: { + schema: { + type: 'object', + properties: {}, + additionalProperties: true, + }, + }, + }) + } }, }, ], diff --git a/executor/handlers.ts b/executor/handlers.ts index 4a3b5e6b47..5ae0b8b888 100644 --- a/executor/handlers.ts +++ b/executor/handlers.ts @@ -480,27 +480,173 @@ export class EvaluatorBlockHandler implements BlockHandler { const model = inputs.model || 'gpt-4o' const providerId = getProviderFromModel(model) - // Parse system prompt object - const systemPromptObj = - typeof inputs.systemPrompt === 'string' - ? JSON.parse(inputs.systemPrompt) - : inputs.systemPrompt + // Process the content to ensure it's in a suitable format + let processedContent = '' - // Execute the evaluator prompt with structured output format + try { + if (typeof inputs.content === 'string') { + if (inputs.content.trim().startsWith('[') || inputs.content.trim().startsWith('{')) { + try { + const parsed = JSON.parse(inputs.content) + processedContent = JSON.stringify(parsed, null, 2) + } catch (e) { + processedContent = inputs.content + } + } else { + processedContent = inputs.content + } + } else if (typeof inputs.content === 'object') { + processedContent = JSON.stringify(inputs.content, null, 2) + } else { + processedContent = String(inputs.content || '') + } + } catch (e) { + console.error('Error processing content:', e) + processedContent = String(inputs.content || '') + } + + // Parse system prompt object with robust error handling + let systemPromptObj: { systemPrompt: string; responseFormat: any } = { + systemPrompt: '', + responseFormat: null, + } + + const metrics = Array.isArray(inputs.metrics) ? inputs.metrics : [] + const metricDescriptions = metrics + .map((m: any) => `"${m.name}" (${m.range.min}-${m.range.max}): ${m.description}`) + .join('\n') + + // Create a response format structure + const responseProperties: Record = {} + metrics.forEach((m: any) => { + responseProperties[m.name] = { type: 'number' } + }) + + systemPromptObj = { + systemPrompt: `You are an evaluation agent. Analyze this content against the metrics and provide scores. + + Metrics: + ${metricDescriptions} + + Content: + ${processedContent} + + Return a JSON object with each metric name as a key and a numeric score as the value. No explanations, only scores.`, + responseFormat: { + name: 'evaluation_response', + schema: { + type: 'object', + properties: responseProperties, + required: metrics.map((m: any) => m.name), + additionalProperties: false, + }, + strict: true, + }, + } + + // Ensure we have a system prompt + if (!systemPromptObj.systemPrompt) { + systemPromptObj.systemPrompt = + 'Evaluate the content and provide scores for each metric as JSON.' + } + + // Make sure we force JSON output in the request const response = await executeProviderRequest(providerId, { model: inputs.model, - systemPrompt: systemPromptObj?.systemPrompt, - responseFormat: systemPromptObj?.responseFormat, - messages: [{ role: 'user', content: inputs.content }], + systemPrompt: systemPromptObj.systemPrompt, + responseFormat: systemPromptObj.responseFormat, + messages: [ + { + role: 'user', + content: + 'Please evaluate the content provided in the system prompt. Return ONLY a valid JSON with metric scores.', + }, + ], temperature: inputs.temperature || 0, apiKey: inputs.apiKey, }) - // Parse response content - const parsedContent = JSON.parse(response.content) + // Parse response content with robust error handling + let parsedContent: Record = {} + try { + const contentStr = response.content.trim() + let jsonStr = '' + + // Method 1: Extract content between first { and last } + const fullMatch = contentStr.match(/(\{[\s\S]*\})/) + if (fullMatch) { + jsonStr = fullMatch[0] + } + // Method 2: Try to find and extract just the JSON part + else if (contentStr.includes('{') && contentStr.includes('}')) { + const startIdx = contentStr.indexOf('{') + const endIdx = contentStr.lastIndexOf('}') + 1 + jsonStr = contentStr.substring(startIdx, endIdx) + } + // Method 3: Just use the raw content as a last resort + else { + jsonStr = contentStr + } + + // Try to parse the extracted JSON + try { + parsedContent = JSON.parse(jsonStr) + } catch (parseError) { + console.error('Failed to parse extracted JSON:', parseError) + throw new Error('Invalid JSON in response') + } + } catch (error) { + console.error('Error parsing evaluator response:', error) + console.error('Raw response content:', response.content) + + // Fallback to empty object + parsedContent = {} + } + + // Extract and process metric scores with proper validation + const metricScores: Record = {} + + try { + const metrics = Array.isArray(inputs.metrics) ? inputs.metrics : [] + + // If we have a successful parse, extract the metrics + if (Object.keys(parsedContent).length > 0) { + metrics.forEach((metric: { name: string }) => { + const metricName = metric.name + + // Try multiple possible ways the metric might be represented + if (parsedContent[metricName] !== undefined) { + metricScores[metricName] = Number(parsedContent[metricName]) + } else if (parsedContent[metricName.toLowerCase()] !== undefined) { + metricScores[metricName] = Number(parsedContent[metricName.toLowerCase()]) + } else if (parsedContent[metricName.toUpperCase()] !== undefined) { + metricScores[metricName] = Number(parsedContent[metricName.toUpperCase()]) + } else { + // Last resort - try to find any key that might contain this metric name + const matchingKey = Object.keys(parsedContent).find((key) => + key.toLowerCase().includes(metricName.toLowerCase()) + ) + + if (matchingKey) { + metricScores[metricName] = Number(parsedContent[matchingKey]) + } else { + console.warn(`Metric "${metricName}" not found in LLM response`) + metricScores[metricName] = 0 + } + } + }) + } else { + // If we couldn't parse any content, set all metrics to 0 + metrics.forEach((metric: { name: string }) => { + metricScores[metric.name] = 0 + }) + } + } catch (e) { + console.error('Error extracting metric scores:', e) + } // Create result with metrics as direct fields for easy access - return { + const result = { response: { content: inputs.content, model: response.model, @@ -509,11 +655,11 @@ export class EvaluatorBlockHandler implements BlockHandler { completion: response.tokens?.completion || 0, total: response.tokens?.total || 0, }, - ...Object.fromEntries( - Object.entries(parsedContent).map(([key, value]) => [key.toLowerCase(), value]) - ), + ...metricScores, }, } + + return result } } diff --git a/providers/index.ts b/providers/index.ts index ad2bd098f8..595b070a10 100644 --- a/providers/index.ts +++ b/providers/index.ts @@ -19,7 +19,12 @@ export async function executeProviderRequest( const structuredOutputInstructions = generateStructuredOutputInstructions( request.responseFormat ) - request.systemPrompt = `${request.systemPrompt}\n\n${structuredOutputInstructions}` + + // Only add additional instructions if they're not empty + if (structuredOutputInstructions.trim()) { + request.systemPrompt = + `${request.systemPrompt || ''}\n\n${structuredOutputInstructions}`.trim() + } } // Execute the request using the provider's implementation diff --git a/providers/utils.ts b/providers/utils.ts index 788dd67c1a..4425a8b96a 100644 --- a/providers/utils.ts +++ b/providers/utils.ts @@ -117,6 +117,13 @@ export function getProviderModels(providerId: ProviderId): string[] { } export function generateStructuredOutputInstructions(responseFormat: any): string { + // If using the new JSON Schema format, don't add additional instructions + // This is necessary because providers now handle the schema directly + if (responseFormat.schema || (responseFormat.type === 'object' && responseFormat.properties)) { + return '' + } + + // Handle legacy format with fields array if (!responseFormat?.fields) return '' function generateFieldStructure(field: any): string { diff --git a/tools/index.ts b/tools/index.ts index 2761a43e35..ee3bb3f17c 100644 --- a/tools/index.ts +++ b/tools/index.ts @@ -589,7 +589,6 @@ async function handleProxyRequest( headers: { 'Content-Type': 'application/json' }, body: JSON.stringify({ toolId, params }), }) - console.log('Proxy response:', response) const result = await response.json()