mirror of
https://github.com/simstudioai/sim.git
synced 2026-09-24 15:45:35 +08:00
fix: update evaluator to align with new responseFormat logic
This commit is contained in:
+105
-18
@@ -27,34 +27,93 @@ interface EvaluatorResponse extends ToolResponse {
|
||||
}
|
||||
|
||||
export const generateEvaluatorPrompt = (metrics: Metric[], content: string): string => {
|
||||
// Create a clear metrics description with name, range, and description
|
||||
const metricsDescription = metrics
|
||||
.map(
|
||||
(metric) => `${metric.name} (${metric.range.min}-${metric.range.max}): ${metric.description}`
|
||||
(metric) =>
|
||||
`"${metric.name}" (${metric.range.min}-${metric.range.max}): ${metric.description}`
|
||||
)
|
||||
.join('\n')
|
||||
|
||||
// Format the content properly - try to detect and format JSON
|
||||
let formattedContent = content
|
||||
try {
|
||||
// If content looks like JSON (starts with { or [)
|
||||
if (
|
||||
typeof content === 'string' &&
|
||||
(content.trim().startsWith('{') || content.trim().startsWith('['))
|
||||
) {
|
||||
// Try to parse and pretty-print
|
||||
const parsedContent = JSON.parse(content)
|
||||
formattedContent = JSON.stringify(parsedContent, null, 2)
|
||||
}
|
||||
// If it's already an object (shouldn't happen here but just in case)
|
||||
else if (typeof content === 'object') {
|
||||
formattedContent = JSON.stringify(content, null, 2)
|
||||
}
|
||||
} catch (e) {
|
||||
console.warn('Warning: Content may not be valid JSON, using as-is', e)
|
||||
formattedContent = content
|
||||
}
|
||||
|
||||
// Generate an example of the expected output format
|
||||
const exampleOutput = metrics.reduce(
|
||||
(acc, metric) => {
|
||||
acc[metric.name] = Math.floor((metric.range.min + metric.range.max) / 2) // Use middle of range as example
|
||||
return acc
|
||||
},
|
||||
{} as Record<string, number>
|
||||
)
|
||||
|
||||
return `You are an objective evaluation agent. Analyze the content against the provided metrics and provide detailed scoring.
|
||||
|
||||
Evaluation Instructions:
|
||||
- You MUST evaluate the content against each metric
|
||||
- For each metric, provide a numeric score within the specified range
|
||||
- Your response must be a valid JSON object with each metric as a number field
|
||||
- Your response MUST be a valid JSON object with each metric name as a key and a numeric score as the value
|
||||
- Use EXACTLY the metric names provided (case-sensitive, no modifications)
|
||||
- Follow the exact schema of the response format provided to you
|
||||
- Do not include explanations in the JSON - only numeric scores
|
||||
- Under any circumstances, do not include any text before or after the JSON object
|
||||
- Do not add any additional fields not specified in the schema
|
||||
- Do not include ANY text before or after the JSON object
|
||||
|
||||
Metrics to evaluate:
|
||||
${metricsDescription}
|
||||
|
||||
Content to evaluate:
|
||||
${content}`
|
||||
${formattedContent}
|
||||
|
||||
Example of expected response format (with different scores):
|
||||
${JSON.stringify(exampleOutput, null, 2)}
|
||||
|
||||
Remember: Your response MUST be a valid JSON object containing only the metrics as keys with their numeric scores as values. No text explanations.`
|
||||
}
|
||||
|
||||
// Simplified response format generator that matches the agent block schema structure
|
||||
const generateResponseFormat = (metrics: Metric[]) => ({
|
||||
fields: metrics.map((metric) => ({
|
||||
name: metric.name,
|
||||
type: 'number',
|
||||
description: `${metric.description} (Score between ${metric.range.min}-${metric.range.max})`,
|
||||
})),
|
||||
})
|
||||
const generateResponseFormat = (metrics: Metric[]) => {
|
||||
// Create properties for each metric
|
||||
const properties: Record<string, any> = {}
|
||||
|
||||
// Add each metric as a property
|
||||
metrics.forEach((metric) => {
|
||||
properties[metric.name] = {
|
||||
type: 'number',
|
||||
description: `${metric.description} (Score between ${metric.range.min}-${metric.range.max})`,
|
||||
}
|
||||
})
|
||||
|
||||
// Return a proper JSON Schema format
|
||||
return {
|
||||
name: 'evaluation_response',
|
||||
schema: {
|
||||
type: 'object',
|
||||
properties,
|
||||
required: metrics.map((metric) => metric.name),
|
||||
additionalProperties: false,
|
||||
},
|
||||
strict: true,
|
||||
}
|
||||
}
|
||||
|
||||
export const EvaluatorBlock: BlockConfig<EvaluatorResponse> = {
|
||||
type: 'evaluator',
|
||||
@@ -102,14 +161,42 @@ export const EvaluatorBlock: BlockConfig<EvaluatorResponse> = {
|
||||
layout: 'full',
|
||||
hidden: true,
|
||||
value: (params: Record<string, any>) => {
|
||||
const metrics = params.metrics || []
|
||||
const content = params.content || ''
|
||||
const responseFormat = generateResponseFormat(metrics)
|
||||
try {
|
||||
const metrics = params.metrics || []
|
||||
|
||||
return JSON.stringify({
|
||||
systemPrompt: generateEvaluatorPrompt(metrics, content),
|
||||
responseFormat,
|
||||
})
|
||||
// Process content safely
|
||||
let processedContent = ''
|
||||
if (typeof params.content === 'object') {
|
||||
processedContent = JSON.stringify(params.content, null, 2)
|
||||
} else {
|
||||
processedContent = String(params.content || '')
|
||||
}
|
||||
|
||||
// Generate prompt and response format directly
|
||||
const promptText = generateEvaluatorPrompt(metrics, processedContent)
|
||||
const responseFormatObj = generateResponseFormat(metrics)
|
||||
|
||||
// Create a clean, simple JSON object
|
||||
const result = {
|
||||
systemPrompt: promptText,
|
||||
responseFormat: responseFormatObj,
|
||||
}
|
||||
|
||||
return JSON.stringify(result)
|
||||
} catch (e) {
|
||||
console.error('Error in systemPrompt value function:', e)
|
||||
// Return a minimal valid JSON as fallback
|
||||
return JSON.stringify({
|
||||
systemPrompt: 'Evaluate the content and return a JSON with metric scores.',
|
||||
responseFormat: {
|
||||
schema: {
|
||||
type: 'object',
|
||||
properties: {},
|
||||
additionalProperties: true,
|
||||
},
|
||||
},
|
||||
})
|
||||
}
|
||||
},
|
||||
},
|
||||
],
|
||||
|
||||
+161
-15
@@ -480,27 +480,173 @@ export class EvaluatorBlockHandler implements BlockHandler {
|
||||
const model = inputs.model || 'gpt-4o'
|
||||
const providerId = getProviderFromModel(model)
|
||||
|
||||
// Parse system prompt object
|
||||
const systemPromptObj =
|
||||
typeof inputs.systemPrompt === 'string'
|
||||
? JSON.parse(inputs.systemPrompt)
|
||||
: inputs.systemPrompt
|
||||
// Process the content to ensure it's in a suitable format
|
||||
let processedContent = ''
|
||||
|
||||
// Execute the evaluator prompt with structured output format
|
||||
try {
|
||||
if (typeof inputs.content === 'string') {
|
||||
if (inputs.content.trim().startsWith('[') || inputs.content.trim().startsWith('{')) {
|
||||
try {
|
||||
const parsed = JSON.parse(inputs.content)
|
||||
processedContent = JSON.stringify(parsed, null, 2)
|
||||
} catch (e) {
|
||||
processedContent = inputs.content
|
||||
}
|
||||
} else {
|
||||
processedContent = inputs.content
|
||||
}
|
||||
} else if (typeof inputs.content === 'object') {
|
||||
processedContent = JSON.stringify(inputs.content, null, 2)
|
||||
} else {
|
||||
processedContent = String(inputs.content || '')
|
||||
}
|
||||
} catch (e) {
|
||||
console.error('Error processing content:', e)
|
||||
processedContent = String(inputs.content || '')
|
||||
}
|
||||
|
||||
// Parse system prompt object with robust error handling
|
||||
let systemPromptObj: { systemPrompt: string; responseFormat: any } = {
|
||||
systemPrompt: '',
|
||||
responseFormat: null,
|
||||
}
|
||||
|
||||
const metrics = Array.isArray(inputs.metrics) ? inputs.metrics : []
|
||||
const metricDescriptions = metrics
|
||||
.map((m: any) => `"${m.name}" (${m.range.min}-${m.range.max}): ${m.description}`)
|
||||
.join('\n')
|
||||
|
||||
// Create a response format structure
|
||||
const responseProperties: Record<string, any> = {}
|
||||
metrics.forEach((m: any) => {
|
||||
responseProperties[m.name] = { type: 'number' }
|
||||
})
|
||||
|
||||
systemPromptObj = {
|
||||
systemPrompt: `You are an evaluation agent. Analyze this content against the metrics and provide scores.
|
||||
|
||||
Metrics:
|
||||
${metricDescriptions}
|
||||
|
||||
Content:
|
||||
${processedContent}
|
||||
|
||||
Return a JSON object with each metric name as a key and a numeric score as the value. No explanations, only scores.`,
|
||||
responseFormat: {
|
||||
name: 'evaluation_response',
|
||||
schema: {
|
||||
type: 'object',
|
||||
properties: responseProperties,
|
||||
required: metrics.map((m: any) => m.name),
|
||||
additionalProperties: false,
|
||||
},
|
||||
strict: true,
|
||||
},
|
||||
}
|
||||
|
||||
// Ensure we have a system prompt
|
||||
if (!systemPromptObj.systemPrompt) {
|
||||
systemPromptObj.systemPrompt =
|
||||
'Evaluate the content and provide scores for each metric as JSON.'
|
||||
}
|
||||
|
||||
// Make sure we force JSON output in the request
|
||||
const response = await executeProviderRequest(providerId, {
|
||||
model: inputs.model,
|
||||
systemPrompt: systemPromptObj?.systemPrompt,
|
||||
responseFormat: systemPromptObj?.responseFormat,
|
||||
messages: [{ role: 'user', content: inputs.content }],
|
||||
systemPrompt: systemPromptObj.systemPrompt,
|
||||
responseFormat: systemPromptObj.responseFormat,
|
||||
messages: [
|
||||
{
|
||||
role: 'user',
|
||||
content:
|
||||
'Please evaluate the content provided in the system prompt. Return ONLY a valid JSON with metric scores.',
|
||||
},
|
||||
],
|
||||
temperature: inputs.temperature || 0,
|
||||
apiKey: inputs.apiKey,
|
||||
})
|
||||
|
||||
// Parse response content
|
||||
const parsedContent = JSON.parse(response.content)
|
||||
// Parse response content with robust error handling
|
||||
let parsedContent: Record<string, any> = {}
|
||||
try {
|
||||
const contentStr = response.content.trim()
|
||||
let jsonStr = ''
|
||||
|
||||
// Method 1: Extract content between first { and last }
|
||||
const fullMatch = contentStr.match(/(\{[\s\S]*\})/)
|
||||
if (fullMatch) {
|
||||
jsonStr = fullMatch[0]
|
||||
}
|
||||
// Method 2: Try to find and extract just the JSON part
|
||||
else if (contentStr.includes('{') && contentStr.includes('}')) {
|
||||
const startIdx = contentStr.indexOf('{')
|
||||
const endIdx = contentStr.lastIndexOf('}') + 1
|
||||
jsonStr = contentStr.substring(startIdx, endIdx)
|
||||
}
|
||||
// Method 3: Just use the raw content as a last resort
|
||||
else {
|
||||
jsonStr = contentStr
|
||||
}
|
||||
|
||||
// Try to parse the extracted JSON
|
||||
try {
|
||||
parsedContent = JSON.parse(jsonStr)
|
||||
} catch (parseError) {
|
||||
console.error('Failed to parse extracted JSON:', parseError)
|
||||
throw new Error('Invalid JSON in response')
|
||||
}
|
||||
} catch (error) {
|
||||
console.error('Error parsing evaluator response:', error)
|
||||
console.error('Raw response content:', response.content)
|
||||
|
||||
// Fallback to empty object
|
||||
parsedContent = {}
|
||||
}
|
||||
|
||||
// Extract and process metric scores with proper validation
|
||||
const metricScores: Record<string, any> = {}
|
||||
|
||||
try {
|
||||
const metrics = Array.isArray(inputs.metrics) ? inputs.metrics : []
|
||||
|
||||
// If we have a successful parse, extract the metrics
|
||||
if (Object.keys(parsedContent).length > 0) {
|
||||
metrics.forEach((metric: { name: string }) => {
|
||||
const metricName = metric.name
|
||||
|
||||
// Try multiple possible ways the metric might be represented
|
||||
if (parsedContent[metricName] !== undefined) {
|
||||
metricScores[metricName] = Number(parsedContent[metricName])
|
||||
} else if (parsedContent[metricName.toLowerCase()] !== undefined) {
|
||||
metricScores[metricName] = Number(parsedContent[metricName.toLowerCase()])
|
||||
} else if (parsedContent[metricName.toUpperCase()] !== undefined) {
|
||||
metricScores[metricName] = Number(parsedContent[metricName.toUpperCase()])
|
||||
} else {
|
||||
// Last resort - try to find any key that might contain this metric name
|
||||
const matchingKey = Object.keys(parsedContent).find((key) =>
|
||||
key.toLowerCase().includes(metricName.toLowerCase())
|
||||
)
|
||||
|
||||
if (matchingKey) {
|
||||
metricScores[metricName] = Number(parsedContent[matchingKey])
|
||||
} else {
|
||||
console.warn(`Metric "${metricName}" not found in LLM response`)
|
||||
metricScores[metricName] = 0
|
||||
}
|
||||
}
|
||||
})
|
||||
} else {
|
||||
// If we couldn't parse any content, set all metrics to 0
|
||||
metrics.forEach((metric: { name: string }) => {
|
||||
metricScores[metric.name] = 0
|
||||
})
|
||||
}
|
||||
} catch (e) {
|
||||
console.error('Error extracting metric scores:', e)
|
||||
}
|
||||
|
||||
// Create result with metrics as direct fields for easy access
|
||||
return {
|
||||
const result = {
|
||||
response: {
|
||||
content: inputs.content,
|
||||
model: response.model,
|
||||
@@ -509,11 +655,11 @@ export class EvaluatorBlockHandler implements BlockHandler {
|
||||
completion: response.tokens?.completion || 0,
|
||||
total: response.tokens?.total || 0,
|
||||
},
|
||||
...Object.fromEntries(
|
||||
Object.entries(parsedContent).map(([key, value]) => [key.toLowerCase(), value])
|
||||
),
|
||||
...metricScores,
|
||||
},
|
||||
}
|
||||
|
||||
return result
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+6
-1
@@ -19,7 +19,12 @@ export async function executeProviderRequest(
|
||||
const structuredOutputInstructions = generateStructuredOutputInstructions(
|
||||
request.responseFormat
|
||||
)
|
||||
request.systemPrompt = `${request.systemPrompt}\n\n${structuredOutputInstructions}`
|
||||
|
||||
// Only add additional instructions if they're not empty
|
||||
if (structuredOutputInstructions.trim()) {
|
||||
request.systemPrompt =
|
||||
`${request.systemPrompt || ''}\n\n${structuredOutputInstructions}`.trim()
|
||||
}
|
||||
}
|
||||
|
||||
// Execute the request using the provider's implementation
|
||||
|
||||
@@ -117,6 +117,13 @@ export function getProviderModels(providerId: ProviderId): string[] {
|
||||
}
|
||||
|
||||
export function generateStructuredOutputInstructions(responseFormat: any): string {
|
||||
// If using the new JSON Schema format, don't add additional instructions
|
||||
// This is necessary because providers now handle the schema directly
|
||||
if (responseFormat.schema || (responseFormat.type === 'object' && responseFormat.properties)) {
|
||||
return ''
|
||||
}
|
||||
|
||||
// Handle legacy format with fields array
|
||||
if (!responseFormat?.fields) return ''
|
||||
|
||||
function generateFieldStructure(field: any): string {
|
||||
|
||||
@@ -589,7 +589,6 @@ async function handleProxyRequest(
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ toolId, params }),
|
||||
})
|
||||
console.log('Proxy response:', response)
|
||||
|
||||
const result = await response.json()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user