From ba3e4cd39fb3adf4a6c99b1bd6c76221279bb740 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Tue, 29 Jul 2025 09:58:05 -0300 Subject: [PATCH] fix: improve robots.txt HTML filtering to check content structure (#1880) - Modify HTML content-type filtering to examine if response body starts with '<' - Allows valid robots.txt content served with text/html content-type headers - Still prevents actual HTML error pages from being parsed as robots.txt - Fixes issue where sites like JPMorgan Chase had robots.txt rules ignored Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: Micah Stairs --- apps/api/src/lib/robots-txt.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/apps/api/src/lib/robots-txt.ts b/apps/api/src/lib/robots-txt.ts index 468850860..0b601d87c 100644 --- a/apps/api/src/lib/robots-txt.ts +++ b/apps/api/src/lib/robots-txt.ts @@ -35,7 +35,7 @@ export async function fetchRobotsTxt( (x) => x[0].toLowerCase() === "content-type", ) ?? [])[1] ?? ""; - if (contentType.includes("text/html") || + if ((contentType.includes("text/html") && response.data.trim().startsWith("<")) || contentType.includes("application/json") || contentType.includes("application/xml")) { return "";