diff --git a/apps/api/src/scraper/scrapeURL/engines/document/index.ts b/apps/api/src/scraper/scrapeURL/engines/document/index.ts index 195acb440..c1352e027 100644 --- a/apps/api/src/scraper/scrapeURL/engines/document/index.ts +++ b/apps/api/src/scraper/scrapeURL/engines/document/index.ts @@ -88,46 +88,46 @@ export async function scrapeDocument(meta: Meta): Promise { let proxyUsed: "basic" | "stealth" = "basic"; let tempFilePath: string | null = null; - if (meta.documentPrefetch !== undefined && meta.documentPrefetch !== null) { - // Use prefetched document - tempFilePath = meta.documentPrefetch.filePath; - buffer = await readFile(tempFilePath); - - // Create a mock response object with content-type from prefetch - const headers = new Headers(); - if (meta.documentPrefetch.contentType) { - headers.set("Content-Type", meta.documentPrefetch.contentType); - } - - response = { - url: meta.documentPrefetch.url ?? meta.rewrittenUrl ?? meta.url, - status: meta.documentPrefetch.status, - headers, - } as Response; - - proxyUsed = meta.documentPrefetch.proxyUsed; - } else { - // Fetch the document normally - const result = await fetchFileToBuffer( - meta.rewrittenUrl ?? meta.url, - meta.options.skipTlsVerification, - { - headers: meta.options.headers, - signal: meta.abort.asSignal(), - }, - ); - response = result.response; - buffer = result.buffer; - - // Validate content type only when fetching directly (not using prefetch) - const ct = response.headers.get("Content-Type"); - if (ct && !isValidDocumentContentType(ct)) { - // if downloaded file wasn't a valid document, throw antibot error - throw new DocumentAntibotError(); - } - } - try { + if (meta.documentPrefetch !== undefined && meta.documentPrefetch !== null) { + // Use prefetched document + tempFilePath = meta.documentPrefetch.filePath; + buffer = await readFile(tempFilePath); + + // Create a mock response object with content-type from prefetch + const headers = new Headers(); + if (meta.documentPrefetch.contentType) { + headers.set("Content-Type", meta.documentPrefetch.contentType); + } + + response = { + url: meta.documentPrefetch.url ?? meta.rewrittenUrl ?? meta.url, + status: meta.documentPrefetch.status, + headers, + } as Response; + + proxyUsed = meta.documentPrefetch.proxyUsed; + } else { + // Fetch the document normally + const result = await fetchFileToBuffer( + meta.rewrittenUrl ?? meta.url, + meta.options.skipTlsVerification, + { + headers: meta.options.headers, + signal: meta.abort.asSignal(), + }, + ); + response = result.response; + buffer = result.buffer; + + // Validate content type only when fetching directly (not using prefetch) + const ct = response.headers.get("Content-Type"); + if (ct && !isValidDocumentContentType(ct)) { + // if downloaded file wasn't a valid document, throw antibot error + throw new DocumentAntibotError(); + } + } + const documentType = getDocumentTypeFromContentType(response.headers.get("content-type")) ?? getDocumentTypeFromUrl(response.url); @@ -145,11 +145,7 @@ export async function scrapeDocument(meta: Meta): Promise { }; } finally { // Clean up temporary file if it was created by prefetch - if ( - tempFilePath && - meta.documentPrefetch !== undefined && - meta.documentPrefetch !== null - ) { + if (tempFilePath) { try { await unlink(tempFilePath); } catch (error) { diff --git a/apps/api/src/scraper/scrapeURL/engines/pdf/index.ts b/apps/api/src/scraper/scrapeURL/engines/pdf/index.ts index e06fa94fc..cf182370c 100644 --- a/apps/api/src/scraper/scrapeURL/engines/pdf/index.ts +++ b/apps/api/src/scraper/scrapeURL/engines/pdf/index.ts @@ -316,125 +316,136 @@ export async function scrapePDF(meta: Meta): Promise { }, ); - if ((response as any).headers) { - // if downloadFile was used - const r: Response = response as any; - const ct = r.headers.get("Content-Type"); - if (ct && !ct.includes("application/pdf")) { - // if downloaded file wasn't a PDF - if (meta.pdfPrefetch === undefined) { - // for non-PDF URLs, this is expected, not anti-bot - if (!meta.featureFlags.has("pdf")) { - throw new EngineUnsuccessfulError("pdf"); + try { + if ((response as any).headers) { + // if downloadFile was used + const r: Response = response as any; + const ct = r.headers.get("Content-Type"); + if (ct && !ct.includes("application/pdf")) { + // if downloaded file wasn't a PDF + if (meta.pdfPrefetch === undefined) { + // for non-PDF URLs, this is expected, not anti-bot + if (!meta.featureFlags.has("pdf")) { + throw new EngineUnsuccessfulError("pdf"); + } else { + throw new PDFAntibotError(); + } } else { - throw new PDFAntibotError(); + throw new PDFPrefetchFailed(); } - } else { - throw new PDFPrefetchFailed(); } } - } - const pdfMetadata = await getPdfMetadata(tempFilePath); - const effectivePageCount = maxPages - ? Math.min(pdfMetadata.numPages, maxPages) - : pdfMetadata.numPages; + const pdfMetadata = await getPdfMetadata(tempFilePath); + const effectivePageCount = maxPages + ? Math.min(pdfMetadata.numPages, maxPages) + : pdfMetadata.numPages; - if ( - effectivePageCount * MILLISECONDS_PER_PAGE > - (meta.abort.scrapeTimeout() ?? Infinity) - ) { - throw new PDFInsufficientTimeError( - effectivePageCount, - effectivePageCount * MILLISECONDS_PER_PAGE + 5000, - ); - } + if ( + effectivePageCount * MILLISECONDS_PER_PAGE > + (meta.abort.scrapeTimeout() ?? Infinity) + ) { + throw new PDFInsufficientTimeError( + effectivePageCount, + effectivePageCount * MILLISECONDS_PER_PAGE + 5000, + ); + } - let result: PDFProcessorResult | null = null; + let result: PDFProcessorResult | null = null; - const base64Content = (await readFile(tempFilePath)).toString("base64"); + const base64Content = (await readFile(tempFilePath)).toString("base64"); - // First try RunPod MU if conditions are met - if ( - base64Content.length < MAX_FILE_SIZE && - process.env.RUNPOD_MU_API_KEY && - process.env.RUNPOD_MU_POD_ID - ) { - const muV1StartedAt = Date.now(); - try { - result = await scrapePDFWithRunPodMU( + // First try RunPod MU if conditions are met + if ( + base64Content.length < MAX_FILE_SIZE && + process.env.RUNPOD_MU_API_KEY && + process.env.RUNPOD_MU_POD_ID + ) { + const muV1StartedAt = Date.now(); + try { + result = await scrapePDFWithRunPodMU( + { + ...meta, + logger: meta.logger.child({ + method: "scrapePDF/scrapePDFWithRunPodMU", + }), + }, + tempFilePath, + base64Content, + maxPages, + ); + const muV1DurationMs = Date.now() - muV1StartedAt; + meta.logger + .child({ method: "scrapePDF/MUv1Experiment" }) + .info("MU v1 completed", { + durationMs: muV1DurationMs, + url: meta.rewrittenUrl ?? meta.url, + pages: effectivePageCount, + success: true, + }); + } catch (error) { + if ( + error instanceof RemoveFeatureError || + error instanceof AbortManagerThrownError + ) { + throw error; + } + meta.logger.warn( + "RunPod MU failed to parse PDF (could be due to timeout) -- falling back to parse-pdf", + { error }, + ); + Sentry.captureException(error); + const muV1DurationMs = Date.now() - muV1StartedAt; + meta.logger + .child({ method: "scrapePDF/MUv1Experiment" }) + .info("MU v1 failed", { + durationMs: muV1DurationMs, + url: meta.rewrittenUrl ?? meta.url, + pages: effectivePageCount, + success: false, + }); + } + } + + // If RunPod MU failed or wasn't attempted, use PdfParse + if (!result) { + result = await scrapePDFWithParsePDF( { ...meta, logger: meta.logger.child({ - method: "scrapePDF/scrapePDFWithRunPodMU", + method: "scrapePDF/scrapePDFWithParsePDF", }), }, tempFilePath, - base64Content, - maxPages, ); - const muV1DurationMs = Date.now() - muV1StartedAt; - meta.logger - .child({ method: "scrapePDF/MUv1Experiment" }) - .info("MU v1 completed", { - durationMs: muV1DurationMs, - url: meta.rewrittenUrl ?? meta.url, - pages: effectivePageCount, - success: true, - }); + } + + return { + url: response.url ?? meta.rewrittenUrl ?? meta.url, + statusCode: response.status, + html: result?.html ?? "", + markdown: result?.markdown ?? "", + pdfMetadata: { + // Rust parser gets the metadata incorrectly, so we overwrite the page count here with the effective page count + // TODO: fix this later + numPages: effectivePageCount, + title: pdfMetadata.title, + }, + + proxyUsed: "basic", + }; + } finally { + // Always clean up temp file after we're done with it + try { + await unlink(tempFilePath); } catch (error) { - if ( - error instanceof RemoveFeatureError || - error instanceof AbortManagerThrownError - ) { - throw error; - } - meta.logger.warn( - "RunPod MU failed to parse PDF (could be due to timeout) -- falling back to parse-pdf", - { error }, - ); - Sentry.captureException(error); - const muV1DurationMs = Date.now() - muV1StartedAt; - meta.logger - .child({ method: "scrapePDF/MUv1Experiment" }) - .info("MU v1 failed", { - durationMs: muV1DurationMs, - url: meta.rewrittenUrl ?? meta.url, - pages: effectivePageCount, - success: false, - }); + // Ignore errors when cleaning up temp files + meta.logger?.warn("Failed to clean up temporary PDF file", { + error, + tempFilePath, + }); } } - - // If RunPod MU failed or wasn't attempted, use PdfParse - if (!result) { - result = await scrapePDFWithParsePDF( - { - ...meta, - logger: meta.logger.child({ - method: "scrapePDF/scrapePDFWithParsePDF", - }), - }, - tempFilePath, - ); - } - - await unlink(tempFilePath); - - return { - url: response.url ?? meta.rewrittenUrl ?? meta.url, - statusCode: response.status, - html: result?.html ?? "", - markdown: result?.markdown ?? "", - pdfMetadata: { - // Rust parser gets the metadata incorrectly, so we overwrite the page count here with the effective page count - // TODO: fix this later - numPages: effectivePageCount, - title: pdfMetadata.title, - }, - - proxyUsed: "basic", - }; } export function pdfMaxReasonableTime(meta: Meta): number { diff --git a/apps/api/src/scraper/scrapeURL/engines/utils/downloadFile.ts b/apps/api/src/scraper/scrapeURL/engines/utils/downloadFile.ts index 66f3d9eb2..ca6c6d37a 100644 --- a/apps/api/src/scraper/scrapeURL/engines/utils/downloadFile.ts +++ b/apps/api/src/scraper/scrapeURL/engines/utils/downloadFile.ts @@ -12,6 +12,7 @@ import { Writable } from "stream"; import { v4 as uuid } from "uuid"; import * as undici from "undici"; import { getSecureDispatcher } from "./safeFetch"; +import { logger } from "../../../../lib/logger"; const mapUndiciError = (url: string, skipTlsVerification: boolean, e: any) => { const code = e?.code ?? e?.cause?.code ?? e?.errno ?? e?.name; @@ -89,6 +90,7 @@ export async function downloadFile( }> { const tempFilePath = path.join(os.tmpdir(), `tempFile-${id}--${uuid()}`); const tempFileWrite = createWriteStream(tempFilePath); + let shouldCleanup = false; // TODO: maybe we could use tlsclient for this? for proxying try { @@ -118,8 +120,20 @@ export async function downloadFile( tempFilePath, }; } catch (e) { + // Mark for cleanup on error (caller handles cleanup on success) + shouldCleanup = true; throw mapUndiciError(url, skipTlsVerification, e); } finally { tempFileWrite.close(); + if (shouldCleanup) { + try { + await fs.unlink(tempFilePath); + } catch (cleanupError: any) { + logger.warn("Failed to clean up temporary file", { + error: cleanupError, + tempFilePath, + }); + } + } } }