diff --git a/.github/workflows/test-server.yml b/.github/workflows/test-server.yml index ca30d1de6..074d1ac79 100644 --- a/.github/workflows/test-server.yml +++ b/.github/workflows/test-server.yml @@ -42,6 +42,7 @@ env: VERTEX_CREDENTIALS: ${{ secrets.VERTEX_CREDENTIALS }} USE_GO_MARKDOWN_PARSER: true SENTRY_ENVIRONMENT: dev + IDMUX_URL: ${{ secrets.IDMUX_URL }} jobs: test: diff --git a/apps/api/src/__tests__/snips/batch-scrape.test.ts b/apps/api/src/__tests__/snips/batch-scrape.test.ts index 9a3f3cba6..3e03b92dc 100644 --- a/apps/api/src/__tests__/snips/batch-scrape.test.ts +++ b/apps/api/src/__tests__/snips/batch-scrape.test.ts @@ -1,10 +1,20 @@ -import { batchScrape, scrapeTimeout } from "./lib"; +import { batchScrape, scrapeTimeout, idmux, Identity } from "./lib"; + +let identity: Identity; + +beforeAll(async () => { + identity = await idmux({ + name: "batch-scrape", + concurrency: 100, + credits: 1000000, + }); +}, 10000); describe("Batch scrape tests", () => { it.concurrent("works", async () => { const response = await batchScrape({ urls: ["http://firecrawl.dev"] - }); + }, identity); expect(response.data[0]).toHaveProperty("markdown"); expect(response.data[0].markdown).toContain("Firecrawl"); @@ -34,7 +44,7 @@ describe("Batch scrape tests", () => { required: ["company_mission", "supports_sso", "is_open_source"], }, }, - }); + }, identity); expect(response.data[0]).toHaveProperty("json"); expect(response.data[0].json).toHaveProperty("company_mission"); @@ -52,7 +62,7 @@ describe("Batch scrape tests", () => { it.concurrent("sourceURL stays unnormalized", async () => { const response = await batchScrape({ urls: ["https://firecrawl.dev/?pagewanted=all&et_blog"], - }); + }, identity); expect(response.data[0].metadata.sourceURL).toBe("https://firecrawl.dev/?pagewanted=all&et_blog"); }, scrapeTimeout); diff --git a/apps/api/src/__tests__/snips/billing.test.ts b/apps/api/src/__tests__/snips/billing.test.ts index 2b54572e0..d1e004438 100644 --- a/apps/api/src/__tests__/snips/billing.test.ts +++ b/apps/api/src/__tests__/snips/billing.test.ts @@ -1,35 +1,33 @@ -import { batchScrape, crawl, creditUsage, extract, map, scrape, search, tokenUsage } from "./lib"; +import { batchScrape, crawl, creditUsage, extract, idmux, map, scrape, search, tokenUsage } from "./lib"; const sleep = (ms: number) => new Promise(x => setTimeout(() => x(true), ms)); const sleepForBatchBilling = () => sleep(40000); -beforeAll(async () => { - // Wait for previous test runs to stop billing processing - if (!process.env.TEST_SUITE_SELF_HOSTED) { - await sleep(40000); - } -}, 50000); - describe("Billing tests", () => { if (process.env.TEST_SUITE_SELF_HOSTED) { it("dummy", () => { expect(true).toBe(true); }); } else { - it("bills scrape correctly", async () => { - const rc1 = (await creditUsage()).remaining_credits; + it.concurrent("bills scrape correctly", async () => { + const identity = await idmux({ + name: "billing/bills scrape correctly", + credits: 100, + }); + + const rc1 = (await creditUsage(identity)).remaining_credits; // Run all scrape operations in parallel with Promise.all const [scrape1, scrape2, scrape3] = await Promise.all([ // scrape 1: regular fc.dev scrape (1 credit) scrape({ url: "https://firecrawl.dev" - }), + }, identity), // scrape 1.1: regular fc.dev scrape (1 credit) scrape({ url: "https://firecrawl.dev" - }), + }, identity), // scrape 2: fc.dev with json (5 credits) scrape({ @@ -44,7 +42,7 @@ describe("Billing tests", () => { required: ["is_open_source"], }, }, - }) + }, identity) ]); expect(scrape1.metadata.creditsUsed).toBe(1); @@ -55,13 +53,18 @@ describe("Billing tests", () => { await sleepForBatchBilling(); - const rc2 = (await creditUsage()).remaining_credits; + const rc2 = (await creditUsage(identity)).remaining_credits; expect(rc1 - rc2).toBe(7); }, 120000); - it("bills batch scrape correctly", async () => { - const rc1 = (await creditUsage()).remaining_credits; + it.concurrent("bills batch scrape correctly", async () => { + const identity = await idmux({ + name: "billing/bills batch scrape correctly", + credits: 100, + }); + + const rc1 = (await creditUsage(identity)).remaining_credits; // Run both scrape operations in parallel with Promise.all const [scrape1, scrape2] = await Promise.all([ @@ -72,7 +75,7 @@ describe("Billing tests", () => { "https://mendable.ai", "https://thisdomaindoesnotexistandwillfail.fcr", ], - }), + }, identity), // scrape 2: batch scrape with json (10 credits) batchScrape({ @@ -91,7 +94,7 @@ describe("Billing tests", () => { required: ["four_word_summary"], }, }, - }) + }, identity) ]); expect(scrape1.data[0].metadata.creditsUsed).toBe(1); @@ -104,13 +107,18 @@ describe("Billing tests", () => { await sleepForBatchBilling(); - const rc2 = (await creditUsage()).remaining_credits; + const rc2 = (await creditUsage(identity)).remaining_credits; expect(rc1 - rc2).toBe(12); }, 600000); - it("bills crawl correctly", async () => { - const rc1 = (await creditUsage()).remaining_credits; + it.concurrent("bills crawl correctly", async () => { + const identity = await idmux({ + name: "billing/bills crawl correctly", + credits: 200, + }); + + const rc1 = (await creditUsage(identity)).remaining_credits; // Run both crawl operations in parallel with Promise.all const [crawl1, crawl2] = await Promise.all([ @@ -118,7 +126,7 @@ describe("Billing tests", () => { crawl({ url: "https://firecrawl.dev", limit: 10, - }), + }, identity), // crawl 2: fc.dev crawl with json (5y credits) crawl({ @@ -136,7 +144,7 @@ describe("Billing tests", () => { }, }, limit: 10, - }) + }, identity) ]); expect(crawl1.success).toBe(true); @@ -146,54 +154,74 @@ describe("Billing tests", () => { await sleepForBatchBilling(); - const rc2 = (await creditUsage()).remaining_credits; + const rc2 = (await creditUsage(identity)).remaining_credits; if (crawl1.success && crawl2.success) { expect(rc1 - rc2).toBe(crawl1.completed + crawl2.completed * 5); } }, 600000); - it("bills map correctly", async () => { - const rc1 = (await creditUsage()).remaining_credits; - await map({ url: "https://firecrawl.dev" }); + it.concurrent("bills map correctly", async () => { + const identity = await idmux({ + name: "billing/bills map correctly", + credits: 100, + }); + + const rc1 = (await creditUsage(identity)).remaining_credits; + await map({ url: "https://firecrawl.dev" }, identity); await sleepForBatchBilling(); - const rc2 = (await creditUsage()).remaining_credits; + const rc2 = (await creditUsage(identity)).remaining_credits; expect(rc1 - rc2).toBe(1); }, 60000); - it("bills search correctly", async () => { - const rc1 = (await creditUsage()).remaining_credits; + it.concurrent("bills search correctly", async () => { + const identity = await idmux({ + name: "billing/bills search correctly", + credits: 100, + }); + + const rc1 = (await creditUsage(identity)).remaining_credits; const results = await search({ query: "firecrawl" - }); + }, identity); await sleepForBatchBilling(); - const rc2 = (await creditUsage()).remaining_credits; + const rc2 = (await creditUsage(identity)).remaining_credits; expect(rc1 - rc2).toBe(results.length); }, 60000); - it("bills search with scrape correctly", async () => { - const rc1 = (await creditUsage()).remaining_credits; + it.concurrent("bills search with scrape correctly", async () => { + const identity = await idmux({ + name: "billing/bills search with scrape correctly", + credits: 100, + }); + + const rc1 = (await creditUsage(identity)).remaining_credits; const results = await search({ query: "firecrawl", scrapeOptions: { formats: ["markdown"], }, - }); + }, identity); await sleepForBatchBilling(); - const rc2 = (await creditUsage()).remaining_credits; + const rc2 = (await creditUsage(identity)).remaining_credits; expect(rc1 - rc2).toBe(results.length); }, 600000); - it("bills extract correctly", async () => { - const rc1 = (await tokenUsage()).remaining_tokens; + it.concurrent("bills extract correctly", async () => { + const identity = await idmux({ + name: "billing/bills extract correctly", + tokens: 1000, + }); + + const rc1 = (await tokenUsage(identity)).remaining_tokens; const extractResult = await extract({ urls: ["https://firecrawl.dev"], @@ -209,13 +237,13 @@ describe("Billing tests", () => { ] }, origin: "api-sdk", - }); + }, identity); expect(extractResult.tokensUsed).toBe(305); await sleepForBatchBilling(); - const rc2 = (await tokenUsage()).remaining_tokens; + const rc2 = (await tokenUsage(identity)).remaining_tokens; expect(rc1 - rc2).toBe(305); }, 300000); diff --git a/apps/api/src/__tests__/snips/concurrency.test.ts b/apps/api/src/__tests__/snips/concurrency.test.ts index 3128145aa..f38f291cd 100644 --- a/apps/api/src/__tests__/snips/concurrency.test.ts +++ b/apps/api/src/__tests__/snips/concurrency.test.ts @@ -1,59 +1,62 @@ -import { batchScrapeWithConcurrencyTracking, concurrencyCheck, crawlWithConcurrencyTracking, defaultIdentity, Identity } from "./lib"; - -const concurrencyIdentity: Identity = { - apiKey: process.env.TEST_API_KEY_CONCURRENCY ?? process.env.TEST_API_KEY!, -} +import { batchScrapeWithConcurrencyTracking, crawlWithConcurrencyTracking, idmux } from "./lib"; if (!process.env.TEST_SUITE_SELF_HOSTED) { - let accountConcurrencyLimit = 2; + const accountConcurrencyLimit = 20; - beforeAll(async () => { - const { maxConcurrency } = await concurrencyCheck(concurrencyIdentity); - accountConcurrencyLimit = maxConcurrency; - - console.log("Account concurrency limit:", accountConcurrencyLimit); - - if (accountConcurrencyLimit > 20) { - console.warn("Your account's concurrency limit (" + accountConcurrencyLimit + ") is likely too high, which will cause these tests to fail. Please set up TEST_API_KEY_CONCURRENCY with an API key that has a lower concurrency limit."); - } - }, 10000); - describe("Concurrency queue and limit", () => { - it("crawl utilizes full concurrency limit and doesn't go over", async () => { + it.concurrent("crawl utilizes full concurrency limit and doesn't go over", async () => { + const identity = await idmux({ + name: "concurrency/crawl utilizes full concurrency limit and doesn't go over", + concurrency: accountConcurrencyLimit, + credits: 100, + }); + const limit = accountConcurrencyLimit * 2; const { crawl, concurrencies } = await crawlWithConcurrencyTracking({ url: "https://firecrawl.dev", limit, - }, concurrencyIdentity); + }, identity); expect(Math.max(...concurrencies)).toBe(accountConcurrencyLimit); expect(crawl.completed).toBe(limit); }, 600000); - it("crawl handles maxConcurrency properly", async () => { + it.concurrent("crawl handles maxConcurrency properly", async () => { + const identity = await idmux({ + name: "concurrency/crawl handles maxConcurrency properly", + concurrency: accountConcurrencyLimit, + credits: 100, + }); + const { crawl, concurrencies } = await crawlWithConcurrencyTracking({ url: "https://firecrawl.dev", limit: 15, maxConcurrency: 5, - }, concurrencyIdentity); + }, identity); expect(Math.max(...concurrencies)).toBe(5); expect(crawl.completed).toBe(15); }, 600000); - it("crawl maxConcurrency stacks properly", async () => { + it.concurrent("crawl maxConcurrency stacks properly", async () => { + const identity = await idmux({ + name: "concurrency/crawl maxConcurrency stacks properly", + concurrency: accountConcurrencyLimit, + credits: 100, + }); + const [{ crawl: crawl1, concurrencies: concurrencies1 }, { crawl: crawl2, concurrencies: concurrencies2 }] = await Promise.all([ crawlWithConcurrencyTracking({ url: "https://firecrawl.dev", limit: 15, maxConcurrency: 5, - }, concurrencyIdentity), + }, identity), crawlWithConcurrencyTracking({ url: "https://firecrawl.dev", limit: 15, maxConcurrency: 5, - }, concurrencyIdentity), + }, identity), ]); expect(Math.max(...concurrencies1, ...concurrencies2)).toBe(10); @@ -61,37 +64,55 @@ if (!process.env.TEST_SUITE_SELF_HOSTED) { expect(crawl2.completed).toBe(15); }, 1200000); - it("batch scrape utilizes full concurrency limit and doesn't go over", async () => { + it.concurrent("batch scrape utilizes full concurrency limit and doesn't go over", async () => { + const identity = await idmux({ + name: "concurrency/batch scrape utilizes full concurrency limit and doesn't go over", + concurrency: accountConcurrencyLimit, + credits: 100, + }); + const limit = accountConcurrencyLimit * 2; const { batchScrape, concurrencies } = await batchScrapeWithConcurrencyTracking({ urls: Array(limit).fill(0).map(_ => `https://firecrawl.dev`), - }, concurrencyIdentity); + }, identity); expect(Math.max(...concurrencies)).toBe(accountConcurrencyLimit); expect(batchScrape.completed).toBe(limit); }, 600000); - it("batch scrape handles maxConcurrency properly", async () => { + it.concurrent("batch scrape handles maxConcurrency properly", async () => { + const identity = await idmux({ + name: "concurrency/batch scrape handles maxConcurrency properly", + concurrency: accountConcurrencyLimit, + credits: 100, + }); + const { batchScrape, concurrencies } = await batchScrapeWithConcurrencyTracking({ urls: Array(15).fill(0).map(_ => `https://firecrawl.dev`), maxConcurrency: 5, - }, concurrencyIdentity); + }, identity); expect(Math.max(...concurrencies)).toBe(5); expect(batchScrape.completed).toBe(15); }, 600000); - it("batch scrape maxConcurrency stacks properly", async () => { + it.concurrent("batch scrape maxConcurrency stacks properly", async () => { + const identity = await idmux({ + name: "concurrency/batch scrape maxConcurrency stacks properly", + concurrency: accountConcurrencyLimit, + credits: 100, + }); + const [{ batchScrape: batchScrape1, concurrencies: concurrencies1 }, { batchScrape: batchScrape2, concurrencies: concurrencies2 }] = await Promise.all([ batchScrapeWithConcurrencyTracking({ urls: Array(15).fill(0).map(_ => `https://firecrawl.dev`), maxConcurrency: 5, - }, concurrencyIdentity), + }, identity), batchScrapeWithConcurrencyTracking({ urls: Array(15).fill(0).map(_ => `https://firecrawl.dev`), maxConcurrency: 5, - }, concurrencyIdentity), + }, identity), ]); expect(Math.max(...concurrencies1, ...concurrencies2)).toBe(10); diff --git a/apps/api/src/__tests__/snips/crawl.test.ts b/apps/api/src/__tests__/snips/crawl.test.ts index 5f57776b9..3a565fc52 100644 --- a/apps/api/src/__tests__/snips/crawl.test.ts +++ b/apps/api/src/__tests__/snips/crawl.test.ts @@ -1,12 +1,22 @@ -import { asyncCrawl, asyncCrawlWaitForFinish, crawl, crawlOngoing, scrapeTimeout } from "./lib"; +import { asyncCrawl, asyncCrawlWaitForFinish, crawl, crawlOngoing, Identity, idmux, scrapeTimeout } from "./lib"; import { describe, it, expect } from "@jest/globals"; +let identity: Identity; + +beforeAll(async () => { + identity = await idmux({ + name: "crawl", + concurrency: 100, + credits: 1000000, + }); +}, 10000); + describe("Crawl tests", () => { it.concurrent("works", async () => { await crawl({ url: "https://firecrawl.dev", limit: 10, - }); + }, identity); }, 10 * scrapeTimeout); it.concurrent("filters URLs properly", async () => { @@ -14,7 +24,7 @@ describe("Crawl tests", () => { url: "https://firecrawl.dev/pricing", includePaths: ["^/pricing$"], limit: 10, - }); + }, identity); expect(res.success).toBe(true); if (res.success) { @@ -32,7 +42,7 @@ describe("Crawl tests", () => { includePaths: ["^https://(www\\.)?firecrawl\\.dev/pricing$"], regexOnFullURL: true, limit: 10, - }); + }, identity); expect(res.success).toBe(true); if (res.success) { @@ -46,22 +56,22 @@ describe("Crawl tests", () => { url: "https://firecrawl.dev", limit: 3, delay: 5, - }); + }, identity); }, 3 * scrapeTimeout + 3 * 5000); it.concurrent("ongoing crawls endpoint works", async () => { const res = await asyncCrawl({ url: "https://firecrawl.dev", limit: 3, - }); + }, identity); - const ongoing = await crawlOngoing(); + const ongoing = await crawlOngoing(identity); expect(ongoing.crawls.find(x => x.id === res.id)).toBeDefined(); - await asyncCrawlWaitForFinish(res.id); + await asyncCrawlWaitForFinish(res.id, identity); - const ongoing2 = await crawlOngoing(); + const ongoing2 = await crawlOngoing(identity); expect(ongoing2.crawls.find(x => x.id === res.id)).toBeUndefined(); }, 3 * scrapeTimeout); @@ -106,7 +116,7 @@ describe("Crawl tests", () => { url: "https://firecrawl.dev", crawlEntireDomain: true, limit: 5, - }); + }, identity); expect(res.success).toBe(true); if (res.success) { @@ -120,7 +130,7 @@ describe("Crawl tests", () => { allowBackwardLinks: false, crawlEntireDomain: true, limit: 5, - }); + }, identity); expect(res.success).toBe(true); if (res.success) { @@ -133,7 +143,7 @@ describe("Crawl tests", () => { url: "https://firecrawl.dev", allowBackwardLinks: true, limit: 5, - }); + }, identity); expect(res.success).toBe(true); if (res.success) { diff --git a/apps/api/src/__tests__/snips/extract.test.ts b/apps/api/src/__tests__/snips/extract.test.ts index b305d3be6..c3e5abeed 100644 --- a/apps/api/src/__tests__/snips/extract.test.ts +++ b/apps/api/src/__tests__/snips/extract.test.ts @@ -1,4 +1,14 @@ -import { extract } from "./lib"; +import { extract, idmux, Identity } from "./lib"; + +let identity: Identity; + +beforeAll(async () => { + identity = await idmux({ + name: "extract", + concurrency: 100, + tokens: 1000000, + }); +}, 10000); describe("Extract tests", () => { if (!process.env.TEST_SUITE_SELF_HOSTED || process.env.OPENAI_API_KEY || process.env.OLLAMA_BASE_URL) { @@ -24,7 +34,7 @@ describe("Extract tests", () => { timeout: 75000, }, origin: "api-sdk", - }); + }, identity); expect(res.data).toHaveProperty("company_mission"); expect(typeof res.data.company_mission).toBe("string") @@ -52,7 +62,7 @@ describe("Extract tests", () => { timeout: 75000, }, origin: "api-sdk", - }); + }, identity); expect(res.data).toHaveProperty("company_name"); expect(typeof res.data.company_name).toBe("string") diff --git a/apps/api/src/__tests__/snips/lib.ts b/apps/api/src/__tests__/snips/lib.ts index 6d0e28fd2..d39db3fc4 100644 --- a/apps/api/src/__tests__/snips/lib.ts +++ b/apps/api/src/__tests__/snips/lib.ts @@ -10,23 +10,70 @@ import request from "supertest"; const TEST_URL = "http://127.0.0.1:3002"; -export type Identity = { - apiKey: string; +// Due to the limited resources of the CI runner, we need to set a longer timeout for the many many scrape tests +export const scrapeTimeout = 90000; +export const indexCooldown = 30000; + +// ========================================= +// idmux +// ========================================= + +export type IdmuxRequest = { + name: string, + + concurrency?: number, + credits?: number, + tokens?: number, + flags?: any, } -export const defaultIdentity: Identity = { - apiKey: process.env.TEST_API_KEY!, -}; +export async function idmux(req: IdmuxRequest): Promise { + if (!process.env.IDMUX_URL) { + if (!process.env.TEST_SUITE_SELF_HOSTED) { + console.warn("IDMUX_URL is not set, using test API key and team ID"); + } + return { + apiKey: process.env.TEST_API_KEY!, + teamId: process.env.TEST_TEAM_ID!, + } + } -// Due to the limited resources of the CI runner, we need to set a longer timeout for the many many scrape tests -export const scrapeTimeout = 75000; -export const indexCooldown = 30000; + let runNumber = parseInt(process.env.GITHUB_RUN_NUMBER!); + if (isNaN(runNumber) || runNumber === null || runNumber === undefined) { + runNumber = 0; + } + + const res = await fetch(process.env.IDMUX_URL + "/", { + method: "POST", + body: JSON.stringify({ + refName: process.env.GITHUB_REF_NAME!, + runNumber, + concurrency: req.concurrency ?? 100, + ...req, + }), + headers: { + "Content-Type": "application/json", + }, + }); + + if (!res.ok) { + console.error(await res.text()); + } + + expect(res.ok).toBe(true); + return await res.json(); +} + +export type Identity = { + apiKey: string; + teamId: string; +} // ========================================= // Scrape API // ========================================= -async function scrapeRaw(body: ScrapeRequestInput, identity = defaultIdentity) { +async function scrapeRaw(body: ScrapeRequestInput, identity: Identity) { return await request(TEST_URL) .post("/v1/scrape") .set("Authorization", `Bearer ${identity.apiKey}`) @@ -46,7 +93,7 @@ function expectScrapeToFail(response: Awaited>) { expect(typeof response.body.error).toBe("string"); } -export async function scrape(body: ScrapeRequestInput, identity = defaultIdentity): Promise { +export async function scrape(body: ScrapeRequestInput, identity: Identity): Promise { const raw = await scrapeRaw(body, identity); expectScrapeToSucceed(raw); if (body.proxy === "stealth") { @@ -57,7 +104,7 @@ export async function scrape(body: ScrapeRequestInput, identity = defaultIdentit return raw.body.data; } -export async function scrapeWithFailure(body: ScrapeRequestInput, identity = defaultIdentity): Promise<{ +export async function scrapeWithFailure(body: ScrapeRequestInput, identity: Identity): Promise<{ success: false; error: string; }> { @@ -66,14 +113,14 @@ export async function scrapeWithFailure(body: ScrapeRequestInput, identity = def return raw.body; } -export async function scrapeStatusRaw(jobId: string, identity = defaultIdentity) { +export async function scrapeStatusRaw(jobId: string, identity: Identity) { return await request(TEST_URL) .get("/v1/scrape/" + encodeURIComponent(jobId)) .set("Authorization", `Bearer ${identity.apiKey}`) .send(); } -export async function scrapeStatus(jobId: string, identity = defaultIdentity): Promise { +export async function scrapeStatus(jobId: string, identity: Identity): Promise { const raw = await scrapeStatusRaw(jobId, identity); expect(raw.statusCode).toBe(200); expect(raw.body.success).toBe(true); @@ -87,7 +134,7 @@ export async function scrapeStatus(jobId: string, identity = defaultIdentity): P // Crawl API // ========================================= -async function crawlStart(body: CrawlRequestInput, identity = defaultIdentity) { +async function crawlStart(body: CrawlRequestInput, identity: Identity) { return await request(TEST_URL) .post("/v1/crawl") .set("Authorization", `Bearer ${identity.apiKey}`) @@ -95,21 +142,21 @@ async function crawlStart(body: CrawlRequestInput, identity = defaultIdentity) { .send(body); } -async function crawlStatus(id: string, identity = defaultIdentity) { +async function crawlStatus(id: string, identity: Identity) { return await request(TEST_URL) .get("/v1/crawl/" + encodeURIComponent(id)) .set("Authorization", `Bearer ${identity.apiKey}`) .send(); } -async function crawlOngoingRaw(identity = defaultIdentity) { +async function crawlOngoingRaw(identity: Identity) { return await request(TEST_URL) .get("/v1/crawl/ongoing") .set("Authorization", `Bearer ${identity.apiKey}`) .send(); } -export async function crawlOngoing(identity = defaultIdentity): Promise> { +export async function crawlOngoing(identity: Identity): Promise> { const res = await crawlOngoingRaw(identity); expect(res.statusCode).toBe(200); expect(res.body.success).toBe(true); @@ -132,13 +179,13 @@ function expectCrawlToSucceed(response: Awaited>) expect(response.body.data.length).toBeGreaterThan(0); } -export async function asyncCrawl(body: CrawlRequestInput, identity = defaultIdentity): Promise> { +export async function asyncCrawl(body: CrawlRequestInput, identity: Identity): Promise> { const cs = await crawlStart(body, identity); expectCrawlStartToSucceed(cs); return cs.body; } -export async function asyncCrawlWaitForFinish(id: string, identity = defaultIdentity): Promise> { +export async function asyncCrawlWaitForFinish(id: string, identity: Identity): Promise> { let x; do { @@ -151,7 +198,7 @@ export async function asyncCrawlWaitForFinish(id: string, identity = defaultIden return x.body; } -export async function crawlErrors(id: string, identity = defaultIdentity): Promise> { +export async function crawlErrors(id: string, identity: Identity): Promise> { const res = await request(TEST_URL) .get("/v1/crawl/" + id + "/errors") .set("Authorization", `Bearer ${identity.apiKey}`) @@ -163,7 +210,7 @@ export async function crawlErrors(id: string, identity = defaultIdentity): Promi return res.body; } -export async function crawl(body: CrawlRequestInput, identity = defaultIdentity): Promise> { +export async function crawl(body: CrawlRequestInput, identity: Identity): Promise> { const cs = await crawlStart(body, identity); expectCrawlStartToSucceed(cs); @@ -188,7 +235,7 @@ export async function crawl(body: CrawlRequestInput, identity = defaultIdentity) // Batch Scrape API // ========================================= -async function batchScrapeStart(body: BatchScrapeRequestInput, identity = defaultIdentity) { +async function batchScrapeStart(body: BatchScrapeRequestInput, identity: Identity) { return await request(TEST_URL) .post("/v1/batch/scrape") .set("Authorization", `Bearer ${identity.apiKey}`) @@ -196,7 +243,7 @@ async function batchScrapeStart(body: BatchScrapeRequestInput, identity = defaul .send(body); } -async function batchScrapeStatus(id: string, identity = defaultIdentity) { +async function batchScrapeStatus(id: string, identity: Identity) { return await request(TEST_URL) .get("/v1/batch/scrape/" + encodeURIComponent(id)) .set("Authorization", `Bearer ${identity.apiKey}`) @@ -219,7 +266,7 @@ function expectBatchScrapeToSucceed(response: Awaited> { +export async function batchScrape(body: BatchScrapeRequestInput, identity: Identity): Promise & { id: string }> { const bss = await batchScrapeStart(body, identity); expectBatchScrapeStartToSucceed(bss); @@ -239,7 +286,7 @@ export async function batchScrape(body: BatchScrapeRequestInput, identity = defa // Map API // ========================================= -export async function map(body: MapRequestInput, identity = defaultIdentity) { +export async function map(body: MapRequestInput, identity: Identity) { return await request(TEST_URL) .post("/v1/map") .set("Authorization", `Bearer ${identity.apiKey}`) @@ -258,7 +305,7 @@ export function expectMapToSucceed(response: Awaited>) { // Extract API // ========================================= -async function extractStart(body: ExtractRequestInput, identity = defaultIdentity) { +async function extractStart(body: ExtractRequestInput, identity: Identity) { return await request(TEST_URL) .post("/v1/extract") .set("Authorization", `Bearer ${identity.apiKey}`) @@ -266,7 +313,7 @@ async function extractStart(body: ExtractRequestInput, identity = defaultIdentit .send(body); } -async function extractStatus(id: string, identity = defaultIdentity) { +async function extractStatus(id: string, identity: Identity) { return await request(TEST_URL) .get("/v1/extract/" + encodeURIComponent(id)) .set("Authorization", `Bearer ${identity.apiKey}`) @@ -288,7 +335,7 @@ function expectExtractToSucceed(response: Awaited { +export async function extract(body: ExtractRequestInput, identity: Identity): Promise { const es = await extractStart(body, identity); expectExtractStartToSucceed(es); @@ -308,7 +355,7 @@ export async function extract(body: ExtractRequestInput, identity = defaultIdent // Search API // ========================================= -async function searchRaw(body: SearchRequestInput, identity = defaultIdentity) { +async function searchRaw(body: SearchRequestInput, identity: Identity) { return await request(TEST_URL) .post("/v1/search") .set("Authorization", `Bearer ${identity.apiKey}`) @@ -324,7 +371,7 @@ function expectSearchToSucceed(response: Awaited>) expect(response.body.data.length).toBeGreaterThan(0); } -export async function search(body: SearchRequestInput, identity = defaultIdentity): Promise { +export async function search(body: SearchRequestInput, identity: Identity): Promise { const raw = await searchRaw(body, identity); expectSearchToSucceed(raw); return raw.body.data; @@ -334,7 +381,7 @@ export async function search(body: SearchRequestInput, identity = defaultIdentit // Billing API // ========================================= -export async function creditUsage(identity = defaultIdentity): Promise<{ remaining_credits: number }> { +export async function creditUsage(identity: Identity): Promise<{ remaining_credits: number }> { const req = (await request(TEST_URL) .get("/v1/team/credit-usage") .set("Authorization", `Bearer ${identity.apiKey}`) @@ -347,7 +394,7 @@ export async function creditUsage(identity = defaultIdentity): Promise<{ remaini return req.body.data; } -export async function tokenUsage(identity = defaultIdentity): Promise<{ remaining_tokens: number }> { +export async function tokenUsage(identity: Identity): Promise<{ remaining_tokens: number }> { return (await request(TEST_URL) .get("/v1/team/token-usage") .set("Authorization", `Bearer ${identity.apiKey}`) @@ -358,7 +405,7 @@ export async function tokenUsage(identity = defaultIdentity): Promise<{ remainin // Concurrency API // ========================================= -export async function concurrencyCheck(identity = defaultIdentity): Promise<{ concurrency: number, maxConcurrency: number }> { +export async function concurrencyCheck(identity: Identity): Promise<{ concurrency: number, maxConcurrency: number }> { const x = (await request(TEST_URL) .get("/v1/concurrency-check") .set("Authorization", `Bearer ${identity.apiKey}`) @@ -369,7 +416,7 @@ export async function concurrencyCheck(identity = defaultIdentity): Promise<{ co return x.body; } -export async function crawlWithConcurrencyTracking(body: CrawlRequestInput, identity = defaultIdentity): Promise<{ +export async function crawlWithConcurrencyTracking(body: CrawlRequestInput, identity: Identity): Promise<{ crawl: Exclude; concurrencies: number[]; }> { @@ -392,7 +439,7 @@ export async function crawlWithConcurrencyTracking(body: CrawlRequestInput, iden }; } -export async function batchScrapeWithConcurrencyTracking(body: BatchScrapeRequestInput, identity = defaultIdentity): Promise<{ +export async function batchScrapeWithConcurrencyTracking(body: BatchScrapeRequestInput, identity: Identity): Promise<{ batchScrape: Exclude; concurrencies: number[]; }> { @@ -428,7 +475,7 @@ async function deepResearchStart(body: { formats?: string[]; topic?: string; jsonOptions?: any; -}, identity = defaultIdentity) { +}, identity: Identity) { return await request(TEST_URL) .post("/v1/deep-research") .set("Authorization", `Bearer ${identity.apiKey}`) @@ -436,7 +483,7 @@ async function deepResearchStart(body: { .send(body); } -async function deepResearchStatus(id: string, identity = defaultIdentity) { +async function deepResearchStatus(id: string, identity: Identity) { return await request(TEST_URL) .get("/v1/deep-research/" + encodeURIComponent(id)) .set("Authorization", `Bearer ${identity.apiKey}`) @@ -459,7 +506,7 @@ export async function deepResearch(body: { formats?: string[]; topic?: string; jsonOptions?: any; -}, identity = defaultIdentity) { +}, identity: Identity) { const ds = await deepResearchStart(body, identity); expectDeepResearchStartToSucceed(ds); diff --git a/apps/api/src/__tests__/snips/map.test.ts b/apps/api/src/__tests__/snips/map.test.ts index 8c9ffe055..304c9dd39 100644 --- a/apps/api/src/__tests__/snips/map.test.ts +++ b/apps/api/src/__tests__/snips/map.test.ts @@ -1,19 +1,29 @@ -import { expectMapToSucceed, map } from "./lib"; +import { expectMapToSucceed, map, idmux, Identity } from "./lib"; + +let identity: Identity; + +beforeAll(async () => { + identity = await idmux({ + name: "map", + concurrency: 100, + credits: 1000000, + }); +}, 10000); describe("Map tests", () => { it.concurrent("basic map succeeds", async () => { const response = await map({ url: "http://firecrawl.dev", - }); + }, identity); expectMapToSucceed(response); - }, 10000); + }, 60000); it.concurrent("times out properly", async () => { const response = await map({ url: "http://firecrawl.dev", timeout: 1 - }); + }, identity); expect(response.statusCode).toBe(408); expect(response.body.success).toBe(false); @@ -25,7 +35,7 @@ describe("Map tests", () => { url: "https://www.hfea.gov.uk", sitemapOnly: true, useMock: "map-query-params", - }); + }, identity); expect(response.statusCode).toBe(200); expect(response.body.success).toBe(true); diff --git a/apps/api/src/__tests__/snips/scrape.test.ts b/apps/api/src/__tests__/snips/scrape.test.ts index 6c8c80d7e..0bcb34c1d 100644 --- a/apps/api/src/__tests__/snips/scrape.test.ts +++ b/apps/api/src/__tests__/snips/scrape.test.ts @@ -1,6 +1,25 @@ -import { scrape, scrapeStatus, scrapeWithFailure, scrapeTimeout, indexCooldown } from "./lib"; +import { scrape, scrapeStatus, scrapeWithFailure, scrapeTimeout, indexCooldown, idmux, Identity } from "./lib"; import crypto from "crypto"; +let identity: Identity; + +beforeAll(async () => { + identity = await idmux({ + name: "scrape", + concurrency: 100, + credits: 1000000, + }); + + if (!process.env.TEST_SUITE_SELF_HOSTED) { + // Needed for change tracking tests to work + await scrape({ + url: "https://example.com", + formats: ["markdown", "changeTracking"], + timeout: scrapeTimeout, + }, identity); + } +}, 10000 + scrapeTimeout); + describe("Scrape tests", () => { it.concurrent("mocking works properly", async () => { // depends on falsified mock mocking-works-properly @@ -11,7 +30,7 @@ describe("Scrape tests", () => { url: "http://firecrawl.dev", useMock: "mocking-works-properly", timeout: scrapeTimeout, - }); + }, identity); expect(response.markdown).toBe( "this is fake data coming from the mocking system!", @@ -22,7 +41,7 @@ describe("Scrape tests", () => { const response = await scrape({ url: "http://firecrawl.dev", timeout: scrapeTimeout, - }); + }, identity); expect(response.markdown).toContain("Firecrawl"); }, scrapeTimeout); @@ -31,7 +50,7 @@ describe("Scrape tests", () => { const response = await scrape({ url: "https://www.rtpro.yamaha.co.jp/RT/docs/misc/kanji-sjis.html", timeout: scrapeTimeout, - }); + }, identity); expect(response.markdown).toContain("ぐ け げ こ ご さ ざ し じ す ず せ ぜ そ ぞ た"); }, scrapeTimeout); @@ -41,7 +60,7 @@ describe("Scrape tests", () => { const response = await scrape({ url: "https://icanhazip.com", timeout: scrapeTimeout, - }); + }, identity); expect(response.markdown?.trim()).toContain(process.env.PROXY_SERVER!.split("://").slice(-1)[0].split(":")[0]); }, scrapeTimeout); @@ -51,7 +70,7 @@ describe("Scrape tests", () => { url: "https://icanhazip.com", waitFor: 100, timeout: scrapeTimeout, - }); + }, identity); expect(response.markdown?.trim()).toContain(process.env.PROXY_SERVER!.split("://").slice(-1)[0].split(":")[0]); }, scrapeTimeout); @@ -63,7 +82,7 @@ describe("Scrape tests", () => { url: "http://firecrawl.dev", waitFor: 2000, timeout: scrapeTimeout, - }); + }, identity); expect(response.markdown).toContain("Firecrawl"); }, scrapeTimeout); @@ -75,7 +94,7 @@ describe("Scrape tests", () => { url: "https://jsonplaceholder.typicode.com/todos/1", formats: ["rawHtml"], timeout: scrapeTimeout, - }); + }, identity); const obj = JSON.parse(response.rawHtml!); expect(obj.id).toBe(1); @@ -87,35 +106,34 @@ describe("Scrape tests", () => { const response = await scrape({ url: "http://firecrawl.dev", timeout: scrapeTimeout, - }); + }, identity); expect(response.markdown).toContain("Firecrawl"); // Give time to propagate to read replica await new Promise(resolve => setTimeout(resolve, 1000)); - const status = await scrapeStatus(response.metadata.scrapeId!); + const status = await scrapeStatus(response.metadata.scrapeId!, identity); expect(JSON.stringify(status)).toBe(JSON.stringify(response)); }, scrapeTimeout); - // describe("Ad blocking (f-e dependant)", () => { - // it.concurrent("blocks ads by default", async () => { - // const response = await scrape({ - // url: "https://www.allrecipes.com/recipe/18185/yum/", - // }); + describe("Ad blocking (f-e dependant)", () => { + it.concurrent("blocking ads works", async () => { + await scrape({ + url: "https://firecrawl.dev", + blockAds: true, + timeout: scrapeTimeout, + }, identity); + }, scrapeTimeout); - // expect(response.markdown).not.toContain(".g.doubleclick.net/"); - // }, 30000); - - // it.concurrent("doesn't block ads if explicitly disabled", async () => { - // const response = await scrape({ - // url: "https://www.allrecipes.com/recipe/18185/yum/", - // blockAds: false, - // }); - - // expect(response.markdown).toMatch(/(\.g\.doubleclick\.net|amazon-adsystem\.com)\//); - // }, 30000); - // }); + it.concurrent("doesn't block ads if explicitly disabled", async () => { + await scrape({ + url: "https://firecrawl.dev", + blockAds: false, + timeout: scrapeTimeout, + }, identity); + }, scrapeTimeout); + }); describe("Index", () => { it.concurrent("caches properly", async () => { @@ -127,7 +145,7 @@ describe("Scrape tests", () => { maxAge: scrapeTimeout * 3, storeInCache: false, timeout: scrapeTimeout, - }); + }, identity); expect(response1.metadata.cacheState).toBe("miss"); @@ -137,7 +155,7 @@ describe("Scrape tests", () => { url, maxAge: scrapeTimeout * 3, timeout: scrapeTimeout, - }); + }, identity); expect(response2.metadata.cacheState).toBe("miss"); @@ -147,7 +165,7 @@ describe("Scrape tests", () => { url, maxAge: scrapeTimeout * 3, timeout: scrapeTimeout, - }); + }, identity); expect(response3.metadata.cacheState).toBe("hit"); expect(response3.metadata.cachedAt).toBeDefined(); @@ -156,7 +174,7 @@ describe("Scrape tests", () => { url, maxAge: 1, timeout: scrapeTimeout, - }); + }, identity); expect(response4.metadata.cacheState).toBe("miss"); }, scrapeTimeout * 4 + 2 * indexCooldown); @@ -169,7 +187,7 @@ describe("Scrape tests", () => { url, timeout: scrapeTimeout, maxAge: scrapeTimeout * 2, - }); + }, identity); expect(response1.metadata.cacheState).toBe("miss"); @@ -179,7 +197,7 @@ describe("Scrape tests", () => { url, timeout: scrapeTimeout, maxAge: scrapeTimeout * 2, - }); + }, identity); expect(response2.metadata.cacheState).toBe("hit"); }, scrapeTimeout * 2 + 2 * indexCooldown); @@ -192,7 +210,7 @@ describe("Scrape tests", () => { url, formats: ["screenshot"], timeout: scrapeTimeout, - }); + }, identity); expect(response1.screenshot).toBeDefined(); @@ -203,7 +221,7 @@ describe("Scrape tests", () => { formats: ["screenshot"], timeout: scrapeTimeout, maxAge: scrapeTimeout * 2, - }); + }, identity); expect(response2.screenshot).toBe(response1.screenshot); @@ -212,7 +230,7 @@ describe("Scrape tests", () => { formats: ["screenshot@fullPage"], timeout: scrapeTimeout, maxAge: scrapeTimeout * 3, - }); + }, identity); expect(response3.screenshot).not.toBe(response1.screenshot); expect(response3.metadata.cacheState).toBe("miss"); @@ -226,7 +244,7 @@ describe("Scrape tests", () => { url, formats: ["screenshot@fullPage"], timeout: scrapeTimeout, - }); + }, identity); expect(response1.screenshot).toBeDefined(); @@ -237,7 +255,7 @@ describe("Scrape tests", () => { formats: ["screenshot@fullPage"], timeout: scrapeTimeout, maxAge: scrapeTimeout * 2, - }); + }, identity); expect(response2.screenshot).toBe(response1.screenshot); @@ -246,7 +264,7 @@ describe("Scrape tests", () => { formats: ["screenshot"], timeout: scrapeTimeout, maxAge: scrapeTimeout * 3, - }); + }, identity); expect(response3.screenshot).not.toBe(response1.screenshot); expect(response3.metadata.cacheState).toBe("miss"); @@ -260,14 +278,14 @@ describe("Scrape tests", () => { url, formats: ["markdown", "changeTracking"], timeout: scrapeTimeout, - }); + }, identity); const response1 = await scrape({ url, formats: ["markdown", "changeTracking"], timeout: scrapeTimeout, maxAge: scrapeTimeout * 2, - }); + }, identity); expect(response1.metadata.cacheState).not.toBeDefined(); @@ -278,7 +296,7 @@ describe("Scrape tests", () => { formats: ["markdown"], timeout: scrapeTimeout, maxAge: scrapeTimeout * 3 + indexCooldown, - }); + }, identity); expect(response2.metadata.cacheState).toBe("hit"); }, scrapeTimeout * 3 + 2 * indexCooldown); @@ -293,7 +311,7 @@ describe("Scrape tests", () => { "X-Test": "test", }, timeout: scrapeTimeout, - }); + }, identity); await new Promise(resolve => setTimeout(resolve, indexCooldown)); @@ -301,7 +319,7 @@ describe("Scrape tests", () => { url, timeout: scrapeTimeout, maxAge: scrapeTimeout * 2 + indexCooldown, - }); + }, identity); expect(response.metadata.cacheState).toBe("miss"); }, scrapeTimeout * 2 + 1 * indexCooldown); @@ -313,7 +331,7 @@ describe("Scrape tests", () => { await scrape({ url, timeout: scrapeTimeout, - }); + }, identity); await new Promise(resolve => setTimeout(resolve, indexCooldown)); @@ -322,7 +340,7 @@ describe("Scrape tests", () => { timeout: scrapeTimeout, maxAge: scrapeTimeout * 2, mobile: true, - }); + }, identity); expect(response1.metadata.cacheState).toBe("miss"); @@ -333,7 +351,7 @@ describe("Scrape tests", () => { timeout: scrapeTimeout, maxAge: scrapeTimeout * 3, mobile: true, - }); + }, identity); expect(response2.metadata.cacheState).toBe("hit"); }, scrapeTimeout * 3 + 2 * indexCooldown); @@ -350,7 +368,7 @@ describe("Scrape tests", () => { "type": "wait", "milliseconds": 1000, }] - }); + }, identity); expect(response1.metadata.cacheState).not.toBeDefined(); @@ -360,7 +378,7 @@ describe("Scrape tests", () => { url, timeout: scrapeTimeout, maxAge: scrapeTimeout * 2, - }); + }, identity); expect(response2.metadata.cacheState).toBe("miss"); }, scrapeTimeout * 2 + 1 * indexCooldown); @@ -372,7 +390,7 @@ describe("Scrape tests", () => { await scrape({ url, timeout: scrapeTimeout, - }); + }, identity); await new Promise(resolve => setTimeout(resolve, indexCooldown)); @@ -381,7 +399,7 @@ describe("Scrape tests", () => { location: { country: "DE" }, maxAge: scrapeTimeout * 2, timeout: scrapeTimeout, - }); + }, identity); expect(response1.metadata.cacheState).toBe("miss"); @@ -392,7 +410,7 @@ describe("Scrape tests", () => { location: { country: "DE" }, timeout: scrapeTimeout, maxAge: scrapeTimeout * 3, - }); + }, identity); expect(response2.metadata.cacheState).toBe("hit"); }, scrapeTimeout * 3 + 2 * indexCooldown); @@ -405,7 +423,7 @@ describe("Scrape tests", () => { url, blockAds: true, timeout: scrapeTimeout, - }); + }, identity); await new Promise(resolve => setTimeout(resolve, indexCooldown)); @@ -414,7 +432,7 @@ describe("Scrape tests", () => { blockAds: true, timeout: scrapeTimeout, maxAge: scrapeTimeout * 2 + indexCooldown, - }); + }, identity); expect(response0.metadata.cacheState).toBe("hit"); @@ -423,7 +441,7 @@ describe("Scrape tests", () => { blockAds: false, timeout: scrapeTimeout, maxAge: scrapeTimeout * 3 + indexCooldown, - }); + }, identity); expect(response1.metadata.cacheState).toBe("miss"); @@ -434,7 +452,7 @@ describe("Scrape tests", () => { blockAds: false, timeout: scrapeTimeout, maxAge: scrapeTimeout * 4 + 2 * indexCooldown, - }); + }, identity); expect(response2.metadata.cacheState).toBe("hit"); }, scrapeTimeout * 4 + 2 * indexCooldown); @@ -448,7 +466,7 @@ describe("Scrape tests", () => { proxy: "stealth", timeout: scrapeTimeout, maxAge: scrapeTimeout, - }); + }, identity); expect(response1.metadata.proxyUsed).toBe("stealth"); expect(response1.metadata.cacheState).not.toBeDefined(); @@ -459,7 +477,7 @@ describe("Scrape tests", () => { url, timeout: scrapeTimeout, maxAge: scrapeTimeout * 2 + indexCooldown, - }); + }, identity); expect(response2.metadata.cacheState).toBe("hit"); @@ -468,19 +486,19 @@ describe("Scrape tests", () => { proxy: "stealth", timeout: scrapeTimeout, maxAge: scrapeTimeout * 3 + indexCooldown, - }); + }, identity); expect(response3.metadata.cacheState).not.toBeDefined(); }, scrapeTimeout * 3 + indexCooldown); it.concurrent("works properly on pages returning 200", async () => { const id = crypto.randomUUID(); - const url = "https://httpstat.us/200?testId=" + id; + const url = "https://firecrawl.dev/?testId=" + id; await scrape({ url, timeout: scrapeTimeout, - }); + }, identity); await new Promise(resolve => setTimeout(resolve, indexCooldown)); @@ -488,7 +506,7 @@ describe("Scrape tests", () => { url, timeout: scrapeTimeout, maxAge: scrapeTimeout * 2, - }); + }, identity); expect(response.metadata.cacheState).toBe("hit"); }, scrapeTimeout * 2 + 1 * indexCooldown); @@ -500,7 +518,7 @@ describe("Scrape tests", () => { url: "https://example.com", formats: ["markdown", "changeTracking"], timeout: scrapeTimeout, - }); + }, identity); expect(response.changeTracking).toBeDefined(); expect(response.changeTracking?.previousScrapeAt).not.toBeNull(); @@ -514,7 +532,7 @@ describe("Scrape tests", () => { modes: ["git-diff"] }, timeout: scrapeTimeout, - }); + }, identity); expect(response.changeTracking).toBeDefined(); expect(response.changeTracking?.previousScrapeAt).not.toBeNull(); @@ -536,7 +554,7 @@ describe("Scrape tests", () => { prompt: "Summarize the changes between the previous and current content", }, timeout: scrapeTimeout, - }); + }, identity); expect(response.changeTracking).toBeDefined(); expect(response.changeTracking?.previousScrapeAt).not.toBeNull(); @@ -570,7 +588,7 @@ describe("Scrape tests", () => { } }, timeout: scrapeTimeout, - }); + }, identity); expect(response.changeTracking).toBeDefined(); expect(response.changeTracking?.previousScrapeAt).not.toBeNull(); @@ -603,7 +621,7 @@ describe("Scrape tests", () => { } }, timeout: scrapeTimeout, - }); + }, identity); expect(response.changeTracking).toBeDefined(); expect(response.changeTracking?.previousScrapeAt).not.toBeNull(); @@ -628,14 +646,14 @@ describe("Scrape tests", () => { formats: ["markdown", "changeTracking"], changeTrackingOptions: { tag: uuid1 }, timeout: scrapeTimeout, - }); + }, identity); const response2 = await scrape({ url: "https://firecrawl.dev/", formats: ["markdown", "changeTracking"], changeTrackingOptions: { tag: uuid2 }, timeout: scrapeTimeout, - }); + }, identity); expect(response1.changeTracking?.previousScrapeAt).toBeNull(); expect(response1.changeTracking?.changeStatus).toBe("new"); @@ -647,7 +665,7 @@ describe("Scrape tests", () => { formats: ["markdown", "changeTracking"], changeTrackingOptions: { tag: uuid1 }, timeout: scrapeTimeout, - }); + }, identity); expect(response3.changeTracking?.previousScrapeAt).not.toBeNull(); expect(response3.changeTracking?.changeStatus).not.toBe("new"); @@ -659,7 +677,7 @@ describe("Scrape tests", () => { await scrape({ url: "https://iplocation.com", timeout: scrapeTimeout, - }); + }, identity); }, scrapeTimeout); it.concurrent("works with country US", async () => { @@ -667,7 +685,7 @@ describe("Scrape tests", () => { url: "https://iplocation.com", location: { country: "US" }, timeout: scrapeTimeout, - }); + }, identity); expect(response.markdown).toContain("| Country | United States |"); }, scrapeTimeout); @@ -679,7 +697,7 @@ describe("Scrape tests", () => { url: "http://firecrawl.dev", formats: ["screenshot"], timeout: scrapeTimeout, - }); + }, identity); expect(typeof response.screenshot).toBe("string"); }, scrapeTimeout); @@ -689,7 +707,7 @@ describe("Scrape tests", () => { url: "http://firecrawl.dev", formats: ["screenshot@fullPage"], timeout: scrapeTimeout, - }); + }, identity); expect(typeof response.screenshot).toBe("string"); }, scrapeTimeout); @@ -701,7 +719,7 @@ describe("Scrape tests", () => { url: "https://firecrawl.dev", timeout: scrapeTimeout, actions: [{ type: "pdf" }], - }); + }, identity); expect(response.actions?.pdfs).toBeDefined(); expect(response.actions?.pdfs?.length).toBe(1); @@ -715,7 +733,7 @@ describe("Scrape tests", () => { await scrape({ url: "http://firecrawl.dev", timeout: scrapeTimeout, - }); + }, identity); }, scrapeTimeout); it.concurrent("basic works", async () => { @@ -723,7 +741,7 @@ describe("Scrape tests", () => { url: "http://firecrawl.dev", proxy: "basic", timeout: scrapeTimeout, - }); + }, identity); }, scrapeTimeout); it.concurrent("stealth works", async () => { @@ -731,7 +749,7 @@ describe("Scrape tests", () => { url: "http://firecrawl.dev", proxy: "stealth", timeout: scrapeTimeout * 2, - }); + }, identity); }, scrapeTimeout * 2); it.concurrent("auto works properly on non-stealth site", async () => { @@ -739,20 +757,21 @@ describe("Scrape tests", () => { url: "http://firecrawl.dev", proxy: "auto", timeout: scrapeTimeout * 2, - }); + }, identity); expect(res.metadata.proxyUsed).toBe("basic"); }, scrapeTimeout * 2); - it.concurrent("auto works properly on 'stealth' site (faked for reliabile testing)", async () => { - const res = await scrape({ - url: "https://httpstat.us/403", - proxy: "auto", - timeout: scrapeTimeout * 2, - }); + // TODO: flaky + // it.concurrent("auto works properly on 'stealth' site (faked for reliabile testing)", async () => { + // const res = await scrape({ + // url: "https://eo16f6718vph4un.m.pipedream.net", // always returns 403 + // proxy: "auto", + // timeout: scrapeTimeout * 2, + // }, identity); - expect(res.metadata.proxyUsed).toBe("stealth"); - }, scrapeTimeout * 2); + // expect(res.metadata.proxyUsed).toBe("stealth"); + // }, scrapeTimeout * 2); }); describe("PDF (f-e dependant)", () => { @@ -769,7 +788,7 @@ describe("Scrape tests", () => { const response = await scrapeWithFailure({ url: "https://ecma-international.org/wp-content/uploads/ECMA-262_15th_edition_june_2024.pdf", timeout: scrapeTimeout, - }); + }, identity); expect(response.error).toContain("Insufficient time to process PDF"); }, scrapeTimeout); @@ -778,7 +797,7 @@ describe("Scrape tests", () => { const response = await scrape({ url: "https://ecma-international.org/wp-content/uploads/ECMA-262_15th_edition_june_2024.pdf", timeout: scrapeTimeout * 5, - }); + }, identity); // text on the last page expect(response.markdown).toContain("Redistribution and use in source and binary forms, with or without modification"); @@ -788,7 +807,7 @@ describe("Scrape tests", () => { const response = await scrape({ url: "https://docs.google.com/document/d/1H-hOLYssS8xXl2o5hxj4ipE7yyhZAX1s7ADYM1Hdlzo/view", timeout: scrapeTimeout * 5, - }); + }, identity); expect(response.markdown).toContain("This is a test to confirm Google Docs scraping abilities."); }, scrapeTimeout * 5); @@ -797,7 +816,7 @@ describe("Scrape tests", () => { const response = await scrape({ url: "https://docs.google.com/presentation/d/1pDKL1UULpr6siq_eVWE1hjqt5MKCgSSuKS_MWahnHAQ/view", timeout: scrapeTimeout * 5, - }); + }, identity); expect(response.markdown).toContain("This is a test to confirm Google Slides scraping abilities."); }, scrapeTimeout * 5); @@ -829,7 +848,7 @@ describe("Scrape tests", () => { }, }, timeout: scrapeTimeout, - }); + }, identity); expect(response).toHaveProperty("json"); expect(response.json).toHaveProperty("company_mission"); @@ -848,7 +867,7 @@ describe("Scrape tests", () => { const response = await scrape({ url: "https://firecrawl.dev/?pagewanted=all&et_blog", timeout: scrapeTimeout, - }); + }, identity); expect(response.metadata.sourceURL).toBe("https://firecrawl.dev/?pagewanted=all&et_blog"); }, scrapeTimeout); @@ -858,7 +877,7 @@ describe("Scrape tests", () => { url: "https://jsonplaceholder.typicode.com/todos/1", formats: ["markdown"], timeout: scrapeTimeout, - }); + }, identity); expect(response.markdown).toContain("```json"); }, scrapeTimeout); diff --git a/apps/api/src/__tests__/snips/search.test.ts b/apps/api/src/__tests__/snips/search.test.ts index 51ce5bf15..1a081a52f 100644 --- a/apps/api/src/__tests__/snips/search.test.ts +++ b/apps/api/src/__tests__/snips/search.test.ts @@ -1,10 +1,20 @@ -import { search } from "./lib"; +import { search, idmux, Identity } from "./lib"; + +let identity: Identity; + +beforeAll(async () => { + identity = await idmux({ + name: "search", + concurrency: 100, + credits: 1000000, + }); +}, 10000); describe("Search tests", () => { it.concurrent("works", async () => { await search({ query: "firecrawl" - }); + }, identity); }, 60000); it.concurrent("works with scrape", async () => { @@ -15,7 +25,7 @@ describe("Search tests", () => { formats: ["markdown"], }, timeout: 120000, - }); + }, identity); for (const doc of res) { expect(doc.markdown).toBeDefined(); diff --git a/apps/api/src/__tests__/snips/webhook.test.ts b/apps/api/src/__tests__/snips/webhook.test.ts index e602d511c..10feb719e 100644 --- a/apps/api/src/__tests__/snips/webhook.test.ts +++ b/apps/api/src/__tests__/snips/webhook.test.ts @@ -1,4 +1,4 @@ -import { crawl, batchScrape } from "./lib"; +import { crawl, batchScrape, idmux, Identity } from "./lib"; import Express from "express"; import bodyParser from "body-parser"; import type { WebhookEventType } from "src/types"; @@ -7,6 +7,16 @@ import type { Document } from "src/controllers/v1/types"; const WEBHOOK_PORT_CRAWL = 3008; const WEBHOOK_PORT_BATCH_SCRAPE = 3009; +let identity: Identity; + +beforeAll(async () => { + identity = await idmux({ + name: "webhook", + concurrency: 100, + credits: 1000000, + }); +}, 10000); + describe("Webhook tests", () => { it.concurrent("webhook works properly for crawl", async () => { const app = Express(); @@ -35,7 +45,7 @@ describe("Webhook tests", () => { webhook: { url: `http://localhost:${WEBHOOK_PORT_CRAWL}/webhook`, }, - }); + }, identity); // wait to settle the webhook calls await new Promise(resolve => setTimeout(resolve, 1000)); @@ -93,7 +103,7 @@ describe("Webhook tests", () => { webhook: { url: `http://localhost:${WEBHOOK_PORT_BATCH_SCRAPE}/webhook`, }, - }); + }, identity); // wait to settle the webhook calls await new Promise(resolve => setTimeout(resolve, 1000));