mirror of
https://github.com/firecrawl/firecrawl.git
synced 2026-09-24 23:10:45 +08:00
JS SDK fixes
This commit is contained in:
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@mendable/firecrawl-js",
|
||||
"version": "3.0.2",
|
||||
"version": "3.0.3",
|
||||
"description": "JavaScript SDK for Firecrawl API",
|
||||
"main": "dist/index.js",
|
||||
"types": "dist/index.d.ts",
|
||||
|
||||
@@ -34,10 +34,12 @@ import type {
|
||||
BatchScrapeResponse,
|
||||
BatchScrapeJob,
|
||||
ExtractResponse,
|
||||
CrawlOptions,
|
||||
BatchScrapeOptions,
|
||||
} from "./types";
|
||||
import { Watcher } from "./watcher";
|
||||
import type { WatcherOptions } from "./watcher";
|
||||
import type { ZodTypeAny, infer as ZodInfer } from "zod";
|
||||
import * as zt from "zod";
|
||||
|
||||
// Helper types to infer the `json` field from a Zod schema included in `formats`
|
||||
type ExtractJsonSchemaFromFormats<Formats> = Formats extends readonly any[]
|
||||
@@ -45,8 +47,8 @@ type ExtractJsonSchemaFromFormats<Formats> = Formats extends readonly any[]
|
||||
: never;
|
||||
|
||||
type InferredJsonFromOptions<Opts> = Opts extends { formats?: infer Fmts }
|
||||
? ExtractJsonSchemaFromFormats<Fmts> extends ZodTypeAny
|
||||
? ZodInfer<ExtractJsonSchemaFromFormats<Fmts>>
|
||||
? ExtractJsonSchemaFromFormats<Fmts> extends zt.ZodTypeAny
|
||||
? zt.infer<ExtractJsonSchemaFromFormats<Fmts>>
|
||||
: unknown
|
||||
: unknown;
|
||||
|
||||
@@ -136,8 +138,8 @@ export class FirecrawlClient {
|
||||
* @param req Crawl configuration (paths, limits, scrapeOptions, webhook, etc.).
|
||||
* @returns Job id and url.
|
||||
*/
|
||||
async startCrawl(url: string, req: Omit<Parameters<typeof startCrawl>[1], "url"> = {}): Promise<CrawlResponse> {
|
||||
return startCrawl(this.http, { url, ...(req as any) });
|
||||
async startCrawl(url: string, req: CrawlOptions = {}): Promise<CrawlResponse> {
|
||||
return startCrawl(this.http, { url, ...req });
|
||||
}
|
||||
/**
|
||||
* Get the status and partial data of a crawl job.
|
||||
@@ -160,8 +162,8 @@ export class FirecrawlClient {
|
||||
* @param req Crawl configuration plus waiter controls (pollInterval, timeout seconds).
|
||||
* @returns Final job snapshot.
|
||||
*/
|
||||
async crawl(url: string, req: Omit<Parameters<typeof startCrawl>[1], "url"> & { pollInterval?: number; timeout?: number } = {}): Promise<CrawlJob> {
|
||||
return crawlWaiter(this.http, { url, ...(req as any) }, req.pollInterval, req.timeout);
|
||||
async crawl(url: string, req: CrawlOptions & { pollInterval?: number; timeout?: number } = {}): Promise<CrawlJob> {
|
||||
return crawlWaiter(this.http, { url, ...req }, req.pollInterval, req.timeout);
|
||||
}
|
||||
/**
|
||||
* Retrieve crawl errors and robots.txt blocks.
|
||||
@@ -192,7 +194,7 @@ export class FirecrawlClient {
|
||||
* @param opts Batch options (scrape options, webhook, concurrency, idempotency key, etc.).
|
||||
* @returns Job id and url.
|
||||
*/
|
||||
async startBatchScrape(urls: string[], opts?: Parameters<typeof startBatchScrape>[2]): Promise<BatchScrapeResponse> {
|
||||
async startBatchScrape(urls: string[], opts?: BatchScrapeOptions): Promise<BatchScrapeResponse> {
|
||||
return startBatchScrape(this.http, urls, opts);
|
||||
}
|
||||
/**
|
||||
@@ -223,7 +225,7 @@ export class FirecrawlClient {
|
||||
* @param opts Batch options plus waiter controls (pollInterval, timeout seconds).
|
||||
* @returns Final job snapshot.
|
||||
*/
|
||||
async batchScrape(urls: string[], opts?: Parameters<typeof startBatchScrape>[2] & { pollInterval?: number; timeout?: number }): Promise<BatchScrapeJob> {
|
||||
async batchScrape(urls: string[], opts?: BatchScrapeOptions & { pollInterval?: number; timeout?: number }): Promise<BatchScrapeJob> {
|
||||
return batchWaiter(this.http, urls, opts);
|
||||
}
|
||||
|
||||
|
||||
@@ -3,24 +3,12 @@ import {
|
||||
type BatchScrapeResponse,
|
||||
type CrawlErrorsResponse,
|
||||
type Document,
|
||||
type ScrapeOptions,
|
||||
type WebhookConfig,
|
||||
type BatchScrapeOptions,
|
||||
} from "../types";
|
||||
import { HttpClient } from "../utils/httpClient";
|
||||
import { ensureValidScrapeOptions } from "../utils/validation";
|
||||
import { normalizeAxiosError, throwForBadResponse } from "../utils/errorHandler";
|
||||
|
||||
export interface StartBatchOptions {
|
||||
options?: ScrapeOptions;
|
||||
webhook?: string | WebhookConfig;
|
||||
appendToId?: string;
|
||||
ignoreInvalidURLs?: boolean;
|
||||
maxConcurrency?: number;
|
||||
zeroDataRetention?: boolean;
|
||||
integration?: string;
|
||||
idempotencyKey?: string;
|
||||
}
|
||||
|
||||
export async function startBatchScrape(
|
||||
http: HttpClient,
|
||||
urls: string[],
|
||||
@@ -33,7 +21,7 @@ export async function startBatchScrape(
|
||||
zeroDataRetention,
|
||||
integration,
|
||||
idempotencyKey,
|
||||
}: StartBatchOptions = {}
|
||||
}: BatchScrapeOptions = {}
|
||||
): Promise<BatchScrapeResponse> {
|
||||
if (!Array.isArray(urls) || urls.length === 0) throw new Error("URLs list cannot be empty");
|
||||
const payload: Record<string, unknown> = { urls };
|
||||
@@ -117,7 +105,7 @@ export async function waitForBatchCompletion(http: HttpClient, jobId: string, po
|
||||
export async function batchScrape(
|
||||
http: HttpClient,
|
||||
urls: string[],
|
||||
opts: StartBatchOptions & { pollInterval?: number; timeout?: number } = {}
|
||||
opts: BatchScrapeOptions & { pollInterval?: number; timeout?: number } = {}
|
||||
): Promise<BatchScrapeJob> {
|
||||
const start = await startBatchScrape(http, urls, opts);
|
||||
return waitForBatchCompletion(http, start.id, opts.pollInterval ?? 2, opts.timeout);
|
||||
|
||||
@@ -4,31 +4,15 @@ import {
|
||||
type CrawlJob,
|
||||
type CrawlResponse,
|
||||
type Document,
|
||||
type ScrapeOptions,
|
||||
type WebhookConfig,
|
||||
type CrawlOptions,
|
||||
} from "../types";
|
||||
import { HttpClient } from "../utils/httpClient";
|
||||
import { ensureValidScrapeOptions } from "../utils/validation";
|
||||
import { normalizeAxiosError, throwForBadResponse } from "../utils/errorHandler";
|
||||
|
||||
export interface CrawlRequest {
|
||||
export type CrawlRequest = CrawlOptions & {
|
||||
url: string;
|
||||
prompt?: string | null;
|
||||
excludePaths?: string[] | null;
|
||||
includePaths?: string[] | null;
|
||||
maxDiscoveryDepth?: number | null;
|
||||
sitemap?: "skip" | "include";
|
||||
ignoreQueryParameters?: boolean;
|
||||
limit?: number | null;
|
||||
crawlEntireDomain?: boolean;
|
||||
allowExternalLinks?: boolean;
|
||||
allowSubdomains?: boolean;
|
||||
delay?: number | null;
|
||||
maxConcurrency?: number | null;
|
||||
webhook?: string | WebhookConfig | null;
|
||||
scrapeOptions?: ScrapeOptions | null;
|
||||
zeroDataRetention?: boolean;
|
||||
}
|
||||
};
|
||||
|
||||
function prepareCrawlPayload(request: CrawlRequest): Record<string, unknown> {
|
||||
if (!request.url || !request.url.trim()) throw new Error("URL cannot be empty");
|
||||
|
||||
@@ -196,6 +196,24 @@ export interface SearchRequest {
|
||||
scrapeOptions?: ScrapeOptions;
|
||||
}
|
||||
|
||||
export interface CrawlOptions {
|
||||
prompt?: string | null;
|
||||
excludePaths?: string[] | null;
|
||||
includePaths?: string[] | null;
|
||||
maxDiscoveryDepth?: number | null;
|
||||
sitemap?: "skip" | "include";
|
||||
ignoreQueryParameters?: boolean;
|
||||
limit?: number | null;
|
||||
crawlEntireDomain?: boolean;
|
||||
allowExternalLinks?: boolean;
|
||||
allowSubdomains?: boolean;
|
||||
delay?: number | null;
|
||||
maxConcurrency?: number | null;
|
||||
webhook?: string | WebhookConfig | null;
|
||||
scrapeOptions?: ScrapeOptions | null;
|
||||
zeroDataRetention?: boolean;
|
||||
}
|
||||
|
||||
export interface CrawlResponse {
|
||||
id: string;
|
||||
url: string;
|
||||
@@ -211,6 +229,17 @@ export interface CrawlJob {
|
||||
data: Document[];
|
||||
}
|
||||
|
||||
export interface BatchScrapeOptions {
|
||||
options?: ScrapeOptions;
|
||||
webhook?: string | WebhookConfig;
|
||||
appendToId?: string;
|
||||
ignoreInvalidURLs?: boolean;
|
||||
maxConcurrency?: number;
|
||||
zeroDataRetention?: boolean;
|
||||
integration?: string;
|
||||
idempotencyKey?: string;
|
||||
}
|
||||
|
||||
export interface BatchScrapeResponse {
|
||||
id: string;
|
||||
url: string;
|
||||
|
||||
Reference in New Issue
Block a user