From 4934b21c668c2ea3040cd5d7d412da3008c2331a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20M=C3=B3ricz?= Date: Tue, 19 Aug 2025 11:27:21 +0200 Subject: [PATCH] JS SDK fixes --- .../{archive => workflows}/publish-js-sdk.yml | 0 apps/js-sdk/firecrawl/package.json | 2 +- apps/js-sdk/firecrawl/src/v2/client.ts | 20 +++++++------ apps/js-sdk/firecrawl/src/v2/methods/batch.ts | 18 ++---------- apps/js-sdk/firecrawl/src/v2/methods/crawl.ts | 22 ++------------ apps/js-sdk/firecrawl/src/v2/types.ts | 29 +++++++++++++++++++ 6 files changed, 47 insertions(+), 44 deletions(-) rename .github/{archive => workflows}/publish-js-sdk.yml (100%) diff --git a/.github/archive/publish-js-sdk.yml b/.github/workflows/publish-js-sdk.yml similarity index 100% rename from .github/archive/publish-js-sdk.yml rename to .github/workflows/publish-js-sdk.yml diff --git a/apps/js-sdk/firecrawl/package.json b/apps/js-sdk/firecrawl/package.json index 985ec7106..ced6199b5 100644 --- a/apps/js-sdk/firecrawl/package.json +++ b/apps/js-sdk/firecrawl/package.json @@ -1,6 +1,6 @@ { "name": "@mendable/firecrawl-js", - "version": "3.0.2", + "version": "3.0.3", "description": "JavaScript SDK for Firecrawl API", "main": "dist/index.js", "types": "dist/index.d.ts", diff --git a/apps/js-sdk/firecrawl/src/v2/client.ts b/apps/js-sdk/firecrawl/src/v2/client.ts index 4e5f3b7dd..7a76ca266 100644 --- a/apps/js-sdk/firecrawl/src/v2/client.ts +++ b/apps/js-sdk/firecrawl/src/v2/client.ts @@ -34,10 +34,12 @@ import type { BatchScrapeResponse, BatchScrapeJob, ExtractResponse, + CrawlOptions, + BatchScrapeOptions, } from "./types"; import { Watcher } from "./watcher"; import type { WatcherOptions } from "./watcher"; -import type { ZodTypeAny, infer as ZodInfer } from "zod"; +import * as zt from "zod"; // Helper types to infer the `json` field from a Zod schema included in `formats` type ExtractJsonSchemaFromFormats = Formats extends readonly any[] @@ -45,8 +47,8 @@ type ExtractJsonSchemaFromFormats = Formats extends readonly any[] : never; type InferredJsonFromOptions = Opts extends { formats?: infer Fmts } - ? ExtractJsonSchemaFromFormats extends ZodTypeAny - ? ZodInfer> + ? ExtractJsonSchemaFromFormats extends zt.ZodTypeAny + ? zt.infer> : unknown : unknown; @@ -136,8 +138,8 @@ export class FirecrawlClient { * @param req Crawl configuration (paths, limits, scrapeOptions, webhook, etc.). * @returns Job id and url. */ - async startCrawl(url: string, req: Omit[1], "url"> = {}): Promise { - return startCrawl(this.http, { url, ...(req as any) }); + async startCrawl(url: string, req: CrawlOptions = {}): Promise { + return startCrawl(this.http, { url, ...req }); } /** * Get the status and partial data of a crawl job. @@ -160,8 +162,8 @@ export class FirecrawlClient { * @param req Crawl configuration plus waiter controls (pollInterval, timeout seconds). * @returns Final job snapshot. */ - async crawl(url: string, req: Omit[1], "url"> & { pollInterval?: number; timeout?: number } = {}): Promise { - return crawlWaiter(this.http, { url, ...(req as any) }, req.pollInterval, req.timeout); + async crawl(url: string, req: CrawlOptions & { pollInterval?: number; timeout?: number } = {}): Promise { + return crawlWaiter(this.http, { url, ...req }, req.pollInterval, req.timeout); } /** * Retrieve crawl errors and robots.txt blocks. @@ -192,7 +194,7 @@ export class FirecrawlClient { * @param opts Batch options (scrape options, webhook, concurrency, idempotency key, etc.). * @returns Job id and url. */ - async startBatchScrape(urls: string[], opts?: Parameters[2]): Promise { + async startBatchScrape(urls: string[], opts?: BatchScrapeOptions): Promise { return startBatchScrape(this.http, urls, opts); } /** @@ -223,7 +225,7 @@ export class FirecrawlClient { * @param opts Batch options plus waiter controls (pollInterval, timeout seconds). * @returns Final job snapshot. */ - async batchScrape(urls: string[], opts?: Parameters[2] & { pollInterval?: number; timeout?: number }): Promise { + async batchScrape(urls: string[], opts?: BatchScrapeOptions & { pollInterval?: number; timeout?: number }): Promise { return batchWaiter(this.http, urls, opts); } diff --git a/apps/js-sdk/firecrawl/src/v2/methods/batch.ts b/apps/js-sdk/firecrawl/src/v2/methods/batch.ts index 796dd54c4..b80ea5305 100644 --- a/apps/js-sdk/firecrawl/src/v2/methods/batch.ts +++ b/apps/js-sdk/firecrawl/src/v2/methods/batch.ts @@ -3,24 +3,12 @@ import { type BatchScrapeResponse, type CrawlErrorsResponse, type Document, - type ScrapeOptions, - type WebhookConfig, + type BatchScrapeOptions, } from "../types"; import { HttpClient } from "../utils/httpClient"; import { ensureValidScrapeOptions } from "../utils/validation"; import { normalizeAxiosError, throwForBadResponse } from "../utils/errorHandler"; -export interface StartBatchOptions { - options?: ScrapeOptions; - webhook?: string | WebhookConfig; - appendToId?: string; - ignoreInvalidURLs?: boolean; - maxConcurrency?: number; - zeroDataRetention?: boolean; - integration?: string; - idempotencyKey?: string; -} - export async function startBatchScrape( http: HttpClient, urls: string[], @@ -33,7 +21,7 @@ export async function startBatchScrape( zeroDataRetention, integration, idempotencyKey, - }: StartBatchOptions = {} + }: BatchScrapeOptions = {} ): Promise { if (!Array.isArray(urls) || urls.length === 0) throw new Error("URLs list cannot be empty"); const payload: Record = { urls }; @@ -117,7 +105,7 @@ export async function waitForBatchCompletion(http: HttpClient, jobId: string, po export async function batchScrape( http: HttpClient, urls: string[], - opts: StartBatchOptions & { pollInterval?: number; timeout?: number } = {} + opts: BatchScrapeOptions & { pollInterval?: number; timeout?: number } = {} ): Promise { const start = await startBatchScrape(http, urls, opts); return waitForBatchCompletion(http, start.id, opts.pollInterval ?? 2, opts.timeout); diff --git a/apps/js-sdk/firecrawl/src/v2/methods/crawl.ts b/apps/js-sdk/firecrawl/src/v2/methods/crawl.ts index 4d9ba3383..d1eb68206 100644 --- a/apps/js-sdk/firecrawl/src/v2/methods/crawl.ts +++ b/apps/js-sdk/firecrawl/src/v2/methods/crawl.ts @@ -4,31 +4,15 @@ import { type CrawlJob, type CrawlResponse, type Document, - type ScrapeOptions, - type WebhookConfig, + type CrawlOptions, } from "../types"; import { HttpClient } from "../utils/httpClient"; import { ensureValidScrapeOptions } from "../utils/validation"; import { normalizeAxiosError, throwForBadResponse } from "../utils/errorHandler"; -export interface CrawlRequest { +export type CrawlRequest = CrawlOptions & { url: string; - prompt?: string | null; - excludePaths?: string[] | null; - includePaths?: string[] | null; - maxDiscoveryDepth?: number | null; - sitemap?: "skip" | "include"; - ignoreQueryParameters?: boolean; - limit?: number | null; - crawlEntireDomain?: boolean; - allowExternalLinks?: boolean; - allowSubdomains?: boolean; - delay?: number | null; - maxConcurrency?: number | null; - webhook?: string | WebhookConfig | null; - scrapeOptions?: ScrapeOptions | null; - zeroDataRetention?: boolean; -} +}; function prepareCrawlPayload(request: CrawlRequest): Record { if (!request.url || !request.url.trim()) throw new Error("URL cannot be empty"); diff --git a/apps/js-sdk/firecrawl/src/v2/types.ts b/apps/js-sdk/firecrawl/src/v2/types.ts index 42b1a7208..838aae7aa 100644 --- a/apps/js-sdk/firecrawl/src/v2/types.ts +++ b/apps/js-sdk/firecrawl/src/v2/types.ts @@ -196,6 +196,24 @@ export interface SearchRequest { scrapeOptions?: ScrapeOptions; } +export interface CrawlOptions { + prompt?: string | null; + excludePaths?: string[] | null; + includePaths?: string[] | null; + maxDiscoveryDepth?: number | null; + sitemap?: "skip" | "include"; + ignoreQueryParameters?: boolean; + limit?: number | null; + crawlEntireDomain?: boolean; + allowExternalLinks?: boolean; + allowSubdomains?: boolean; + delay?: number | null; + maxConcurrency?: number | null; + webhook?: string | WebhookConfig | null; + scrapeOptions?: ScrapeOptions | null; + zeroDataRetention?: boolean; +} + export interface CrawlResponse { id: string; url: string; @@ -211,6 +229,17 @@ export interface CrawlJob { data: Document[]; } +export interface BatchScrapeOptions { + options?: ScrapeOptions; + webhook?: string | WebhookConfig; + appendToId?: string; + ignoreInvalidURLs?: boolean; + maxConcurrency?: number; + zeroDataRetention?: boolean; + integration?: string; + idempotencyKey?: string; +} + export interface BatchScrapeResponse { id: string; url: string;