From 4cc0b4ed20a6b9200505a6a74d0c19ffb5a434a4 Mon Sep 17 00:00:00 2001 From: tom <62763456+amplitudesxd@users.noreply.github.com> Date: Tue, 16 Sep 2025 20:46:55 +0100 Subject: [PATCH] fix(scrapeURL): handle non-PDF URL errors (#2147) --- .../src/scraper/scrapeURL/engines/pdf/index.ts | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/apps/api/src/scraper/scrapeURL/engines/pdf/index.ts b/apps/api/src/scraper/scrapeURL/engines/pdf/index.ts index c34b5d6fa..0f6442e53 100644 --- a/apps/api/src/scraper/scrapeURL/engines/pdf/index.ts +++ b/apps/api/src/scraper/scrapeURL/engines/pdf/index.ts @@ -12,6 +12,7 @@ import { PDFInsufficientTimeError, PDFPrefetchFailed, RemoveFeatureError, + EngineUnsuccessfulError, } from "../../error"; import { readFile, unlink } from "node:fs/promises"; import path from "node:path"; @@ -200,7 +201,12 @@ export async function scrapePDF(meta: Meta): Promise { if (ct && !ct.includes("application/pdf")) { // if downloaded file wasn't a PDF if (meta.pdfPrefetch === undefined) { - throw new PDFAntibotError(); + // for non-PDF URLs, this is expected, not anti-bot + if (!meta.featureFlags.has("pdf")) { + throw new EngineUnsuccessfulError("pdf"); + } else { + throw new PDFAntibotError(); + } } else { throw new PDFPrefetchFailed(); } @@ -233,7 +239,12 @@ export async function scrapePDF(meta: Meta): Promise { if (ct && !ct.includes("application/pdf")) { // if downloaded file wasn't a PDF if (meta.pdfPrefetch === undefined) { - throw new PDFAntibotError(); + // for non-PDF URLs, this is expected, not anti-bot + if (!meta.featureFlags.has("pdf")) { + throw new EngineUnsuccessfulError("pdf"); + } else { + throw new PDFAntibotError(); + } } else { throw new PDFPrefetchFailed(); }