From d89c64a186c12c5b2a10f241cf81ec895b1bc888 Mon Sep 17 00:00:00 2001 From: Leonardo Grigorio <48296347+leonardogrig@users.noreply.github.com> Date: Thu, 18 Dec 2025 14:54:42 -0300 Subject: [PATCH 1/3] Add firecrawl agent and agent status tools Introduces two new tools: 'firecrawl_agent' for autonomous web data gathering based on a prompt, and 'firecrawl_agent_status' to check the status of agent jobs. Also updates @mendable/firecrawl-js dependency to version 4.9.3. --- package-lock.json | 12 ++--- package.json | 2 +- src/index.ts | 119 ++++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 126 insertions(+), 7 deletions(-) diff --git a/package-lock.json b/package-lock.json index e5b8737..9c33287 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,15 +1,15 @@ { "name": "firecrawl-mcp", - "version": "3.6.0", + "version": "3.6.2", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "firecrawl-mcp", - "version": "3.6.0", + "version": "3.6.2", "license": "MIT", "dependencies": { - "@mendable/firecrawl-js": "^4.3.6", + "@mendable/firecrawl-js": "^4.9.3", "dotenv": "^17.2.2", "firecrawl-fastmcp": "^1.0.4", "typescript": "^5.9.2", @@ -36,9 +36,9 @@ } }, "node_modules/@mendable/firecrawl-js": { - "version": "4.3.6", - "resolved": "https://registry.npmjs.org/@mendable/firecrawl-js/-/firecrawl-js-4.3.6.tgz", - "integrity": "sha512-gha1nICp+yJJt/fjLj1EsptTvlYdjkJRa66clEV+R2F5hJlDOWOjdGAPUU7IO/RXJ+coR1JmxUJ3P6R/ziPyLg==", + "version": "4.9.3", + "resolved": "https://registry.npmjs.org/@mendable/firecrawl-js/-/firecrawl-js-4.9.3.tgz", + "integrity": "sha512-1k6qv0RiFHanx1XQE+DqEjdaQk0IXbsz/MF7FFrHCQX/oPHXm3TtA5gNNvUIogfX1mghgkVthKObmBNoUhVB1Q==", "license": "MIT", "dependencies": { "axios": "^1.12.2", diff --git a/package.json b/package.json index aa69008..00de5e9 100644 --- a/package.json +++ b/package.json @@ -28,7 +28,7 @@ }, "license": "MIT", "dependencies": { - "@mendable/firecrawl-js": "^4.3.6", + "@mendable/firecrawl-js": "^4.9.3", "dotenv": "^17.2.2", "firecrawl-fastmcp": "^1.0.4", "typescript": "^5.9.2", diff --git a/src/index.ts b/src/index.ts index 26b182c..f1de046 100644 --- a/src/index.ts +++ b/src/index.ts @@ -616,6 +616,125 @@ Extract structured information from web pages using LLM capabilities. Supports b return asText(res); }, }); + +server.addTool({ + name: 'firecrawl_agent', + description: ` +Autonomous web data gathering agent. Describe what data you want, and the agent searches, navigates, and extracts it from anywhere on the web. + +**Best for:** Complex data gathering tasks where you don't know the exact URLs; research tasks requiring multiple sources; finding data in hard-to-reach places. +**Not recommended for:** Simple single-page scraping (use scrape); when you already know the exact URL (use scrape or extract). +**Key advantages over extract:** +- No URLs required - just describe what you need +- Autonomously searches and navigates the web +- Faster and more cost-effective for complex tasks +- Higher reliability for varied queries + +**Arguments:** +- prompt: Natural language description of the data you want (required, max 10,000 characters) +- urls: Optional array of URLs to focus the agent on specific pages +- schema: Optional JSON schema for structured output + +**Prompt Example:** "Find the founders of Firecrawl and their backgrounds" +**Usage Example (no URLs):** +\`\`\`json +{ + "name": "firecrawl_agent", + "arguments": { + "prompt": "Find the top 5 AI startups founded in 2024 and their funding amounts", + "schema": { + "type": "object", + "properties": { + "startups": { + "type": "array", + "items": { + "type": "object", + "properties": { + "name": { "type": "string" }, + "funding": { "type": "string" }, + "founded": { "type": "string" } + } + } + } + } + } + } +} +\`\`\` +**Usage Example (with URLs):** +\`\`\`json +{ + "name": "firecrawl_agent", + "arguments": { + "urls": ["https://docs.firecrawl.dev", "https://firecrawl.dev/pricing"], + "prompt": "Compare the features and pricing information from these pages" + } +} +\`\`\` +**Returns:** Extracted data matching your prompt/schema, plus credits used. +`, + parameters: z.object({ + prompt: z.string().min(1).max(10000), + urls: z.array(z.string().url()).optional(), + schema: z.record(z.string(), z.any()).optional(), + }), + execute: async ( + args: unknown, + { session, log }: { session?: SessionData; log: Logger } + ): Promise => { + const client = getClient(session); + const a = args as Record; + log.info('Starting agent', { + prompt: (a.prompt as string).substring(0, 100), + urlCount: Array.isArray(a.urls) ? a.urls.length : 0, + }); + const agentBody = removeEmptyTopLevel({ + prompt: a.prompt as string, + urls: a.urls as string[] | undefined, + schema: (a.schema as Record) || undefined, + }); + const res = await (client as any).agent({ + ...agentBody, + origin: ORIGIN, + }); + return asText(res); + }, +}); + +server.addTool({ + name: 'firecrawl_agent_status', + description: ` +Check the status of an agent job. + +**Usage Example:** +\`\`\`json +{ + "name": "firecrawl_agent_status", + "arguments": { + "id": "550e8400-e29b-41d4-a716-446655440000" + } +} +\`\`\` +**Possible statuses:** +- processing: Agent is still working +- completed: Extraction finished successfully +- failed: An error occurred + +**Returns:** Status, progress, and results (if completed) of the agent job. +`, + parameters: z.object({ id: z.string() }), + execute: async ( + args: unknown, + { session, log }: { session?: SessionData; log: Logger } + ): Promise => { + const client = getClient(session); + const { id } = args as { id: string }; + log.info('Checking agent status', { id }); + const res = await (client as any).getAgentStatus(id); + return asText(res); + }, +}); + const PORT = Number(process.env.PORT || 3000); const HOST = process.env.CLOUD_SERVICE === 'true' From a9b6bf7101603937ea6196263737cdecdd47f50c Mon Sep 17 00:00:00 2001 From: Leonardo Grigorio <48296347+leonardogrig@users.noreply.github.com> Date: Thu, 18 Dec 2025 15:18:10 -0300 Subject: [PATCH 2/3] Add enterprise options to search parameters Introduces an optional 'enterprise' parameter to the search API, allowing users to specify 'zdr' for zero data retention or 'anon' for anonymous mode. Updates documentation to describe these new options. --- src/index.ts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/index.ts b/src/index.ts index f1de046..5cc33c2 100644 --- a/src/index.ts +++ b/src/index.ts @@ -418,6 +418,7 @@ The query also supports search operators, that you can use if needed to refine t } \`\`\` **Returns:** Array of search results (with optional scraped content). +**Enterprise:** Use \`enterprise: ["zdr"]\` for zero data retention (no logging, 5x credits) or \`["anon"]\` for anonymous mode (no logging). `, parameters: z.object({ query: z.string().min(1), @@ -429,6 +430,7 @@ The query also supports search operators, that you can use if needed to refine t .array(z.object({ type: z.enum(['web', 'images', 'news']) })) .optional(), scrapeOptions: scrapeParamsSchema.omit({ url: true }).partial().optional(), + enterprise: z.array(z.enum(['default', 'anon', 'zdr'])).optional(), }), execute: async ( args: unknown, From 3421a797ca86b2e6a0aba7ef002f2b04cd6358d8 Mon Sep 17 00:00:00 2001 From: Leonardo Grigorio <48296347+leonardogrig@users.noreply.github.com> Date: Thu, 18 Dec 2025 15:44:17 -0300 Subject: [PATCH 3/3] Update index.ts --- src/index.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/index.ts b/src/index.ts index 5cc33c2..6e32c3b 100644 --- a/src/index.ts +++ b/src/index.ts @@ -256,6 +256,7 @@ const scrapeParamsSchema = z.object({ }) .optional(), storeInCache: z.boolean().optional(), + zeroDataRetention: z.boolean().optional(), maxAge: z.number().optional(), }); @@ -418,7 +419,6 @@ The query also supports search operators, that you can use if needed to refine t } \`\`\` **Returns:** Array of search results (with optional scraped content). -**Enterprise:** Use \`enterprise: ["zdr"]\` for zero data retention (no logging, 5x credits) or \`["anon"]\` for anonymous mode (no logging). `, parameters: z.object({ query: z.string().min(1),