JS SDK fixes

This commit is contained in:
Gergő Móricz
2025-08-19 11:27:21 +02:00
parent 1f21ec32ba
commit 4934b21c66
6 changed files with 47 additions and 44 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@mendable/firecrawl-js",
"version": "3.0.2",
"version": "3.0.3",
"description": "JavaScript SDK for Firecrawl API",
"main": "dist/index.js",
"types": "dist/index.d.ts",
+11 -9
View File
@@ -34,10 +34,12 @@ import type {
BatchScrapeResponse,
BatchScrapeJob,
ExtractResponse,
CrawlOptions,
BatchScrapeOptions,
} from "./types";
import { Watcher } from "./watcher";
import type { WatcherOptions } from "./watcher";
import type { ZodTypeAny, infer as ZodInfer } from "zod";
import * as zt from "zod";
// Helper types to infer the `json` field from a Zod schema included in `formats`
type ExtractJsonSchemaFromFormats<Formats> = Formats extends readonly any[]
@@ -45,8 +47,8 @@ type ExtractJsonSchemaFromFormats<Formats> = Formats extends readonly any[]
: never;
type InferredJsonFromOptions<Opts> = Opts extends { formats?: infer Fmts }
? ExtractJsonSchemaFromFormats<Fmts> extends ZodTypeAny
? ZodInfer<ExtractJsonSchemaFromFormats<Fmts>>
? ExtractJsonSchemaFromFormats<Fmts> extends zt.ZodTypeAny
? zt.infer<ExtractJsonSchemaFromFormats<Fmts>>
: unknown
: unknown;
@@ -136,8 +138,8 @@ export class FirecrawlClient {
* @param req Crawl configuration (paths, limits, scrapeOptions, webhook, etc.).
* @returns Job id and url.
*/
async startCrawl(url: string, req: Omit<Parameters<typeof startCrawl>[1], "url"> = {}): Promise<CrawlResponse> {
return startCrawl(this.http, { url, ...(req as any) });
async startCrawl(url: string, req: CrawlOptions = {}): Promise<CrawlResponse> {
return startCrawl(this.http, { url, ...req });
}
/**
* Get the status and partial data of a crawl job.
@@ -160,8 +162,8 @@ export class FirecrawlClient {
* @param req Crawl configuration plus waiter controls (pollInterval, timeout seconds).
* @returns Final job snapshot.
*/
async crawl(url: string, req: Omit<Parameters<typeof startCrawl>[1], "url"> & { pollInterval?: number; timeout?: number } = {}): Promise<CrawlJob> {
return crawlWaiter(this.http, { url, ...(req as any) }, req.pollInterval, req.timeout);
async crawl(url: string, req: CrawlOptions & { pollInterval?: number; timeout?: number } = {}): Promise<CrawlJob> {
return crawlWaiter(this.http, { url, ...req }, req.pollInterval, req.timeout);
}
/**
* Retrieve crawl errors and robots.txt blocks.
@@ -192,7 +194,7 @@ export class FirecrawlClient {
* @param opts Batch options (scrape options, webhook, concurrency, idempotency key, etc.).
* @returns Job id and url.
*/
async startBatchScrape(urls: string[], opts?: Parameters<typeof startBatchScrape>[2]): Promise<BatchScrapeResponse> {
async startBatchScrape(urls: string[], opts?: BatchScrapeOptions): Promise<BatchScrapeResponse> {
return startBatchScrape(this.http, urls, opts);
}
/**
@@ -223,7 +225,7 @@ export class FirecrawlClient {
* @param opts Batch options plus waiter controls (pollInterval, timeout seconds).
* @returns Final job snapshot.
*/
async batchScrape(urls: string[], opts?: Parameters<typeof startBatchScrape>[2] & { pollInterval?: number; timeout?: number }): Promise<BatchScrapeJob> {
async batchScrape(urls: string[], opts?: BatchScrapeOptions & { pollInterval?: number; timeout?: number }): Promise<BatchScrapeJob> {
return batchWaiter(this.http, urls, opts);
}
+3 -15
View File
@@ -3,24 +3,12 @@ import {
type BatchScrapeResponse,
type CrawlErrorsResponse,
type Document,
type ScrapeOptions,
type WebhookConfig,
type BatchScrapeOptions,
} from "../types";
import { HttpClient } from "../utils/httpClient";
import { ensureValidScrapeOptions } from "../utils/validation";
import { normalizeAxiosError, throwForBadResponse } from "../utils/errorHandler";
export interface StartBatchOptions {
options?: ScrapeOptions;
webhook?: string | WebhookConfig;
appendToId?: string;
ignoreInvalidURLs?: boolean;
maxConcurrency?: number;
zeroDataRetention?: boolean;
integration?: string;
idempotencyKey?: string;
}
export async function startBatchScrape(
http: HttpClient,
urls: string[],
@@ -33,7 +21,7 @@ export async function startBatchScrape(
zeroDataRetention,
integration,
idempotencyKey,
}: StartBatchOptions = {}
}: BatchScrapeOptions = {}
): Promise<BatchScrapeResponse> {
if (!Array.isArray(urls) || urls.length === 0) throw new Error("URLs list cannot be empty");
const payload: Record<string, unknown> = { urls };
@@ -117,7 +105,7 @@ export async function waitForBatchCompletion(http: HttpClient, jobId: string, po
export async function batchScrape(
http: HttpClient,
urls: string[],
opts: StartBatchOptions & { pollInterval?: number; timeout?: number } = {}
opts: BatchScrapeOptions & { pollInterval?: number; timeout?: number } = {}
): Promise<BatchScrapeJob> {
const start = await startBatchScrape(http, urls, opts);
return waitForBatchCompletion(http, start.id, opts.pollInterval ?? 2, opts.timeout);
+3 -19
View File
@@ -4,31 +4,15 @@ import {
type CrawlJob,
type CrawlResponse,
type Document,
type ScrapeOptions,
type WebhookConfig,
type CrawlOptions,
} from "../types";
import { HttpClient } from "../utils/httpClient";
import { ensureValidScrapeOptions } from "../utils/validation";
import { normalizeAxiosError, throwForBadResponse } from "../utils/errorHandler";
export interface CrawlRequest {
export type CrawlRequest = CrawlOptions & {
url: string;
prompt?: string | null;
excludePaths?: string[] | null;
includePaths?: string[] | null;
maxDiscoveryDepth?: number | null;
sitemap?: "skip" | "include";
ignoreQueryParameters?: boolean;
limit?: number | null;
crawlEntireDomain?: boolean;
allowExternalLinks?: boolean;
allowSubdomains?: boolean;
delay?: number | null;
maxConcurrency?: number | null;
webhook?: string | WebhookConfig | null;
scrapeOptions?: ScrapeOptions | null;
zeroDataRetention?: boolean;
}
};
function prepareCrawlPayload(request: CrawlRequest): Record<string, unknown> {
if (!request.url || !request.url.trim()) throw new Error("URL cannot be empty");
+29
View File
@@ -196,6 +196,24 @@ export interface SearchRequest {
scrapeOptions?: ScrapeOptions;
}
export interface CrawlOptions {
prompt?: string | null;
excludePaths?: string[] | null;
includePaths?: string[] | null;
maxDiscoveryDepth?: number | null;
sitemap?: "skip" | "include";
ignoreQueryParameters?: boolean;
limit?: number | null;
crawlEntireDomain?: boolean;
allowExternalLinks?: boolean;
allowSubdomains?: boolean;
delay?: number | null;
maxConcurrency?: number | null;
webhook?: string | WebhookConfig | null;
scrapeOptions?: ScrapeOptions | null;
zeroDataRetention?: boolean;
}
export interface CrawlResponse {
id: string;
url: string;
@@ -211,6 +229,17 @@ export interface CrawlJob {
data: Document[];
}
export interface BatchScrapeOptions {
options?: ScrapeOptions;
webhook?: string | WebhookConfig;
appendToId?: string;
ignoreInvalidURLs?: boolean;
maxConcurrency?: number;
zeroDataRetention?: boolean;
integration?: string;
idempotencyKey?: string;
}
export interface BatchScrapeResponse {
id: string;
url: string;