diff --git a/apps/api/src/__tests__/snips/v2/crawl.test.ts b/apps/api/src/__tests__/snips/v2/crawl.test.ts index 419635c75..cfe4e2abf 100644 --- a/apps/api/src/__tests__/snips/v2/crawl.test.ts +++ b/apps/api/src/__tests__/snips/v2/crawl.test.ts @@ -389,7 +389,7 @@ describe("Crawl tests", () => { }); } - it.only( + it.concurrent( "shows warning when robots.txt blocks URLs", async () => { // Test with a site that has robots.txt blocking some paths @@ -400,14 +400,51 @@ describe("Crawl tests", () => { ignoreRobotsTxt: false, // Respect robots.txt }, identity, + false, // Don't expect to succeed (robots.txt might block everything) ); - // Check if warning is present when robots.txt blocks URLs - if (results.warning) { + expect(results.success).toBe(true); + expect(results.status).toBe("completed"); + + // Check specifically for robots.txt warning + if (results.warning && results.warning.includes("robots.txt")) { expect(results.warning).toContain("robots.txt"); expect(results.warning).toContain("/scrape endpoint"); } }, 10 * scrapeTimeout, ); + + it.concurrent( + "shows warning when crawl results ≤ 1 and URL is not base domain", + async () => { + // Test with a specific path that should return few results + const results = await crawl( + { + url: "https://mairistumpf.com/some/specific/path", + limit: 10, + ignoreRobotsTxt: false, + }, + identity, + false, // Don't expect to succeed (might get limitedresults) + ); + + expect(results.success).toBe(true); + expect(results.status).toBe("completed"); + + // Check specifically for crawl results warning + if ( + results.warning && + results.warning.includes("Only") && + results.warning.includes("result(s) found") + ) { + expect(results.warning).toContain("Only"); + expect(results.warning).toContain("result(s) found"); + expect(results.warning).toContain("crawlEntireDomain=true"); + expect(results.warning).toContain("higher-level path"); + expect(results.warning).toContain("mairistumpf.com"); + } + }, + 10 * scrapeTimeout, + ); }); diff --git a/apps/api/src/controllers/v2/crawl-status.ts b/apps/api/src/controllers/v2/crawl-status.ts index 6d7172da4..d62edb503 100644 --- a/apps/api/src/controllers/v2/crawl-status.ts +++ b/apps/api/src/controllers/v2/crawl-status.ts @@ -24,6 +24,7 @@ import { getJobFromGCS } from "../../lib/gcs-jobs"; import { scrapeQueue, NuQJob, NuQJobStatus } from "../../services/worker/nuq"; import { ScrapeJobSingleUrls } from "../../types"; import { redisEvictConnection } from "../../services/redis"; +import { isBaseDomain, extractBaseDomain } from "../../lib/url-utils"; configDotenv(); export type PseudoJob = { @@ -380,7 +381,7 @@ export async function crawlStatusController( scrapes.push(scrape); bytes += JSON.stringify(scrape).length; } else { - logger.warn("Job was considered done, but returnvalue is undefined!", { + logger.warn("Job was considered done, but return value is undefined!", { jobId: id, returnvalue: scrape, }); @@ -431,7 +432,7 @@ export async function crawlStatusController( bytes += JSON.stringify(job.returnvalue).length; } else { logger.warn( - "Job was considered done, but returnvalue is undefined!", + "Job was considered done, but return value is undefined!", { scrapeId: job.id, crawlId: req.params.jobId, @@ -483,6 +484,24 @@ export async function crawlStatusController( logger.debug("Failed to check robots blocked URLs", { error }); } + // Check if we should warn about base domain for crawl results + const resultCount = outputBulkB.data.length; + if (!warning && resultCount <= 1) { + // Get the original crawl URL and options from stored crawl data + const crawl = await getCrawl(req.params.jobId); + if (crawl && crawl.originUrl && !isBaseDomain(crawl.originUrl)) { + // Don't show warning if user is already using crawlEntireDomain + const isUsingCrawlEntireDomain = + crawl.crawlerOptions?.crawlEntireDomain === true; + if (!isUsingCrawlEntireDomain) { + const baseDomain = extractBaseDomain(crawl.originUrl); + if (baseDomain) { + warning = `Only ${resultCount} result(s) found. For broader coverage, try crawling with crawlEntireDomain=true or start from a higher-level path like ${baseDomain}`; + } + } + } + } + return res.status(200).json({ success: true, status: outputBulkA.status ?? "scraping",