diff --git a/apps/api/sharedLibs/html-transformer/src/lib.rs b/apps/api/sharedLibs/html-transformer/src/lib.rs index ff65a4978..a93fa2619 100644 --- a/apps/api/sharedLibs/html-transformer/src/lib.rs +++ b/apps/api/sharedLibs/html-transformer/src/lib.rs @@ -533,6 +533,195 @@ pub unsafe extern "C" fn get_inner_json(html: *const libc::c_char) -> *mut libc: CString::new(out).unwrap().into_raw() } +fn _extract_images(html: &str, base_url: &str) -> Result, Box> { + let document = parse_html().one(html); + let base_url = Url::parse(base_url)?; + let base_href = _extract_base_href_from_document(&document, &base_url)?; + let base_href_url = Url::parse(&base_href)?; + let mut images = HashSet::::new(); + + // Helper function to resolve image URLs + let resolve_image_url = |src: &str| -> Result> { + // Skip data URIs and blob URLs + if src.starts_with("data:") || src.starts_with("blob:") { + return Ok(src.to_string()); + } + + // Handle absolute URLs + if src.starts_with("http://") || src.starts_with("https://") { + return Ok(src.to_string()); + } + + // Handle protocol-relative URLs + if src.starts_with("//") { + let resolved = base_url.join(src)?; + return Ok(resolved.to_string()); + } + + // Handle relative URLs + let resolved = base_href_url.join(src)?; + Ok(resolved.to_string()) + }; + + // Extract from img tags + let img_elements: Vec<_> = match document.select("img").map_err(|_| "Failed to select img tags") { + Ok(x) => x.collect(), + Err(e) => return Err(e.into()), + }; + + for img in img_elements { + let attrs = img.attributes.borrow(); + + // Extract from src attribute + if let Some(src) = attrs.get("src") { + if let Ok(resolved) = resolve_image_url(src) { + images.insert(resolved); + } + } + + // Extract from data-src (lazy loading) + if let Some(data_src) = attrs.get("data-src") { + if let Ok(resolved) = resolve_image_url(data_src) { + images.insert(resolved); + } + } + + // Extract from srcset (responsive images) + if let Some(srcset) = attrs.get("srcset") { + for part in srcset.split(',') { + if let Some(url) = part.trim().split_whitespace().next() { + if !url.is_empty() { + if let Ok(resolved) = resolve_image_url(url) { + images.insert(resolved); + } + } + } + } + } + } + + // Extract from picture source elements + let source_elements: Vec<_> = match document.select("picture source").map_err(|_| "Failed to select picture source") { + Ok(x) => x.collect(), + Err(_) => Vec::new(), + }; + + for source in source_elements { + if let Some(srcset) = source.attributes.borrow().get("srcset") { + for part in srcset.split(',') { + if let Some(url) = part.trim().split_whitespace().next() { + if !url.is_empty() { + if let Ok(resolved) = resolve_image_url(url) { + images.insert(resolved); + } + } + } + } + } + } + + // Extract from meta tags (Open Graph, Twitter Cards) + let meta_selectors = [ + "meta[property=\"og:image\"]", + "meta[property=\"og:image:url\"]", + "meta[property=\"og:image:secure_url\"]", + "meta[name=\"twitter:image\"]", + "meta[name=\"twitter:image:src\"]", + "meta[itemprop=\"image\"]" + ]; + + for selector in &meta_selectors { + if let Ok(elements) = document.select(selector) { + for element in elements { + if let Some(content) = element.attributes.borrow().get("content") { + if let Ok(resolved) = resolve_image_url(content) { + images.insert(resolved); + } + } + } + } + } + + // Extract from link tags (favicons, apple-touch-icons) + let link_selectors = [ + "link[rel*=\"icon\"]", + "link[rel*=\"apple-touch-icon\"]", + "link[rel*=\"image_src\"]" + ]; + + for selector in &link_selectors { + if let Ok(elements) = document.select(selector) { + for element in elements { + if let Some(href) = element.attributes.borrow().get("href") { + if let Ok(resolved) = resolve_image_url(href) { + images.insert(resolved); + } + } + } + } + } + + // Extract from video poster attributes + if let Ok(video_elements) = document.select("video[poster]") { + for video in video_elements { + if let Some(poster) = video.attributes.borrow().get("poster") { + if let Ok(resolved) = resolve_image_url(poster) { + images.insert(resolved); + } + } + } + } + + // Filter out javascript: URLs for security and validate URLs + let filtered_images: Vec = images.into_iter() + .filter(|url| !url.to_lowercase().starts_with("javascript:")) + .filter(|url| !url.is_empty()) + .filter(|url| { + // Validate URLs (allow data URIs and valid URLs) + url.starts_with("data:") || url.starts_with("blob:") || Url::parse(url).is_ok() + }) + .collect(); + + Ok(filtered_images) +} + +/// Extracts images from HTML +/// +/// # Safety +/// Input must be a C HTML string and base URL string. Output will be a JSON string array. Output string must be freed with free_string. +#[no_mangle] +pub unsafe extern "C" fn extract_images(html: *const libc::c_char, base_url: *const libc::c_char) -> *mut libc::c_char { + let html = match unsafe { CStr::from_ptr(html) }.to_str().map_err(|_| ()) { + Ok(x) => x, + Err(_) => { + return CString::new("RUSTFC:ERROR:Failed to parse input HTML as C string").unwrap().into_raw(); + } + }; + + let base_url = match unsafe { CStr::from_ptr(base_url) }.to_str().map_err(|_| ()) { + Ok(x) => x, + Err(_) => { + return CString::new("RUSTFC:ERROR:Failed to parse input base URL as C string").unwrap().into_raw(); + } + }; + + let images = match _extract_images(html, base_url) { + Ok(x) => x, + Err(e) => { + return CString::new(format!("RUSTFC:ERROR:{}", e)).unwrap().into_raw(); + } + }; + + let images_out = match serde_json::ser::to_string(&images) { + Ok(x) => x, + Err(e) => { + return CString::new(format!("RUSTFC:ERROR:{}", e)).unwrap().into_raw(); + } + }; + + CString::new(images_out).unwrap().into_raw() +} + /// Frees a string allocated in Rust-land. /// /// # Safety diff --git a/apps/api/src/__tests__/snips/v2/scrape.test.ts b/apps/api/src/__tests__/snips/v2/scrape.test.ts index 55761395b..da70a777b 100644 --- a/apps/api/src/__tests__/snips/v2/scrape.test.ts +++ b/apps/api/src/__tests__/snips/v2/scrape.test.ts @@ -120,6 +120,39 @@ describe("Scrape tests", () => { expect(response.links?.length).toBeGreaterThan(0); }); + it.concurrent("images format works", async () => { + const response = await scrape({ + url: "https://firecrawl.dev", + formats: ["images"], + }, identity); + + expect(response.images).toBeDefined(); + expect(response.images?.length).toBeGreaterThan(0); + // Firecrawl website should have at least the logo + expect(response.images?.some(img => img.includes("firecrawl"))).toBe(true); + }); + + it.concurrent("images format works with multiple formats", async () => { + const response = await scrape({ + url: "https://firecrawl.dev", + formats: ["markdown", "links", "images"], + }, identity); + + expect(response.markdown).toBeDefined(); + expect(response.links).toBeDefined(); + expect(response.images).toBeDefined(); + expect(response.images?.length).toBeGreaterThan(0); + + // Images should include things that aren't in links + const imageExtensions = ['.jpg', '.jpeg', '.png', '.gif', '.webp', '.svg', '.ico']; + const linkImages = response.links?.filter(link => + imageExtensions.some(ext => link.toLowerCase().includes(ext)) + ) || []; + + // Should have found more images than just those with obvious extensions in links + expect(response.images?.length).toBeGreaterThanOrEqual(linkImages.length); + }); + if (process.env.TEST_SUITE_SELF_HOSTED && process.env.PROXY_SERVER) { it.concurrent("self-hosted proxy works", async () => { const response = await scrape({ diff --git a/apps/api/src/controllers/v1/types.ts b/apps/api/src/controllers/v1/types.ts index 9cee77a61..2e0055888 100644 --- a/apps/api/src/controllers/v1/types.ts +++ b/apps/api/src/controllers/v1/types.ts @@ -829,6 +829,7 @@ export type Document = { html?: string; rawHtml?: string; links?: string[]; + images?: string[]; screenshot?: string; extract?: any; json?: any; diff --git a/apps/api/src/controllers/v2/types.ts b/apps/api/src/controllers/v2/types.ts index fa2c8b225..d7222bf01 100644 --- a/apps/api/src/controllers/v2/types.ts +++ b/apps/api/src/controllers/v2/types.ts @@ -35,6 +35,7 @@ export type Format = | "html" | "rawHtml" | "links" + | "images" | "screenshot" | "screenshot@fullPage" | "extract" @@ -216,6 +217,7 @@ export type FormatObject = | { type: "html" } | { type: "rawHtml" } | { type: "links" } + | { type: "images" } | { type: "summary" } | JsonFormatWithOptions | ChangeTrackingFormatWithOptions @@ -250,6 +252,7 @@ const baseScrapeOptions = z z.object({ type: z.literal("html") }), z.object({ type: z.literal("rawHtml") }), z.object({ type: z.literal("links") }), + z.object({ type: z.literal("images") }), z.object({ type: z.literal("summary") }), jsonFormatWithOptions, changeTrackingFormatWithOptions, @@ -629,6 +632,7 @@ export type Document = { html?: string; rawHtml?: string; links?: string[]; + images?: string[]; screenshot?: string; extract?: any; json?: any; @@ -1349,6 +1353,7 @@ export const searchRequestSchema = z z.object({ type: z.literal("html") }), z.object({ type: z.literal("rawHtml") }), z.object({ type: z.literal("links") }), + z.object({ type: z.literal("images") }), z.object({ type: z.literal("summary") }), jsonFormatWithOptions, screenshotFormatWithOptions, diff --git a/apps/api/src/lib/html-transformer.ts b/apps/api/src/lib/html-transformer.ts index ce334292f..68bf6eb4e 100644 --- a/apps/api/src/lib/html-transformer.ts +++ b/apps/api/src/lib/html-transformer.ts @@ -22,6 +22,7 @@ type TransformHtmlOptions = { class RustHTMLTransformer { private static instance: RustHTMLTransformer; private _extractLinks: KoffiFunction; + private _extractImages: KoffiFunction; private _extractBaseHref: KoffiFunction; private _extractMetadata: KoffiFunction; private _transformHtml: KoffiFunction; @@ -34,6 +35,7 @@ class RustHTMLTransformer { const cstn = "CString:" + crypto.randomUUID(); const freedResultString = koffi.disposable(cstn, "string", this._freeString); this._extractLinks = lib.func("extract_links", freedResultString, ["string"]); + this._extractImages = lib.func("extract_images", freedResultString, ["string", "string"]); this._extractBaseHref = lib.func("extract_base_href", freedResultString, ["string", "string"]); this._extractMetadata = lib.func("extract_metadata", freedResultString, ["string"]); this._transformHtml = lib.func("transform_html", freedResultString, ["string"]); @@ -64,6 +66,22 @@ class RustHTMLTransformer { }); } + public async extractImages(html: string, baseUrl: string): Promise { + return new Promise((resolve, reject) => { + this._extractImages.async(html, baseUrl, (err: Error, res: string) => { + if (err) { + reject(err); + } else { + if (res.startsWith("RUSTFC:ERROR:")) { + reject(new Error(res.replace("RUSTFC:ERROR:", ""))); + } else { + resolve(JSON.parse(res)); + } + } + }); + }); + } + public async extractBaseHref(html: string, url: string): Promise { return new Promise((resolve, reject) => { this._extractBaseHref.async(html, url, (err: Error, res: string) => { @@ -132,6 +150,18 @@ export async function extractLinks( return await converter.extractLinks(html); } +export async function extractImages( + html: string | null | undefined, + baseUrl: string = '' +): Promise { + if (!html) { + return []; + } + + const converter = await RustHTMLTransformer.getInstance(); + return await converter.extractImages(html, baseUrl); +} + export async function extractBaseHref( html: string | null | undefined, url: string diff --git a/apps/api/src/scraper/scrapeURL/lib/__tests__/extractImages.test.ts b/apps/api/src/scraper/scrapeURL/lib/__tests__/extractImages.test.ts new file mode 100644 index 000000000..b8ad9887f --- /dev/null +++ b/apps/api/src/scraper/scrapeURL/lib/__tests__/extractImages.test.ts @@ -0,0 +1,231 @@ +import { extractImages } from '../extractImages'; + +describe('extractImages', () => { + const baseUrl = 'https://example.com/page.html'; + + it('should extract images from img tags', async () => { + const html = ` + + + Test image 1 + Test image 2 + External image + + + `; + + const images = await extractImages(html, baseUrl); + + expect(images).toContain('https://example.com/image1.jpg'); + expect(images).toContain('https://example.com/images/image2.png'); + expect(images).toContain('https://external.com/image3.gif'); + expect(images).toHaveLength(3); + }); + + it('should extract lazy-loaded images from data-src', async () => { + const html = ` + + + Lazy loaded image + Both src and data-src + + + `; + + const images = await extractImages(html, baseUrl); + + expect(images).toContain('https://example.com/lazy-image.webp'); + expect(images).toContain('https://example.com/regular.jpg'); + expect(images).toContain('https://example.com/ignored.jpg'); + }); + + it('should extract images from srcset', async () => { + const html = ` + + + Responsive image + + + `; + + const images = await extractImages(html, baseUrl); + + expect(images).toContain('https://example.com/small.jpg'); + expect(images).toContain('https://example.com/medium.jpg'); + expect(images).toContain('https://example.com/large.jpg'); + expect(images).toHaveLength(3); + }); + + it('should extract images from picture elements', async () => { + const html = ` + + + + + + Picture element + + + + `; + + const images = await extractImages(html, baseUrl); + + expect(images).toContain('https://example.com/image.avif'); + expect(images).toContain('https://example.com/image.webp'); + expect(images).toContain('https://example.com/image.jpg'); + }); + + it('should extract images from meta tags', async () => { + const html = ` + + + + + + + + + `; + + const images = await extractImages(html, baseUrl); + + expect(images).toContain('https://example.com/og-image.jpg'); + expect(images).toContain('https://example.com/og-image-secure.jpg'); + expect(images).toContain('https://example.com/twitter-image.png'); + expect(images).toContain('https://example.com/schema-image.jpg'); + }); + + it('should extract images from link tags', async () => { + const html = ` + + + + + + + + `; + + const images = await extractImages(html, baseUrl); + + expect(images).toContain('https://example.com/favicon.ico'); + expect(images).toContain('https://example.com/apple-touch-icon.png'); + expect(images).toContain('https://example.com/link-image.jpg'); + }); + + it('should extract background images from inline styles', async () => { + const html = ` + + +
Content
+
Content
+
Content
+ + + `; + + const images = await extractImages(html, baseUrl); + + expect(images).toContain('https://example.com/background1.jpg'); + expect(images).toContain('https://example.com/images/background2.png'); + expect(images).toContain('https://example.com/background3.gif'); + }); + + it('should extract video poster images', async () => { + const html = ` + + + + + + + `; + + const images = await extractImages(html, baseUrl); + + expect(images).toContain('https://example.com/video-poster.jpg'); + expect(images).toContain('https://example.com/videos/poster.png'); + }); + + it('should handle protocol-relative URLs', async () => { + const html = ` + + + Protocol relative + + + `; + + const images = await extractImages(html, baseUrl); + + expect(images).toContain('https://cdn.example.com/image.jpg'); + }); + + it('should handle data URIs', async () => { + const html = ` + + + Data URI + + + `; + + const images = await extractImages(html, baseUrl); + + expect(images).toContain('data:image/png;base64,iVBORw0KGgoAAAANS...'); + }); + + it('should respect base tag', async () => { + const html = ` + + + + + + Image with base + + + `; + + const images = await extractImages(html, baseUrl); + + expect(images).toContain('https://different.com/base/image.jpg'); + }); + + it('should remove duplicates', async () => { + const html = ` + + + First + Second + Third + + + `; + + const images = await extractImages(html, baseUrl); + + const duplicateCount = images.filter(img => img.includes('duplicate.jpg')).length; + expect(duplicateCount).toBe(1); + }); + + it('should handle invalid URLs gracefully', async () => { + const html = ` + + + Valid + Invalid + Empty + No src + + + `; + + const images = await extractImages(html, baseUrl); + + expect(images).toContain('https://example.com/valid.jpg'); + expect(images).not.toContain("javascript:alert('xss')"); + expect(images).toHaveLength(1); + }); +}); diff --git a/apps/api/src/scraper/scrapeURL/lib/extractImages.ts b/apps/api/src/scraper/scrapeURL/lib/extractImages.ts new file mode 100644 index 000000000..80f509256 --- /dev/null +++ b/apps/api/src/scraper/scrapeURL/lib/extractImages.ts @@ -0,0 +1,193 @@ +import { load } from "cheerio"; +import { logger } from "../../../lib/logger"; +import { extractImages as _extractImages } from "../../../lib/html-transformer"; + +function resolveImageUrl(src: string, baseUrl: string, baseHref: string = ''): string { + let resolutionBase = baseUrl; + + if (baseHref) { + try { + new URL(baseHref); + resolutionBase = baseHref; + } catch { + try { + resolutionBase = new URL(baseHref, baseUrl).href; + } catch { + resolutionBase = baseUrl; + } + } + } + + try { + // Skip data URIs and blob URLs + if (src.startsWith("data:") || src.startsWith("blob:")) { + return src; + } + + // Handle absolute URLs + if (src.startsWith("http://") || src.startsWith("https://")) { + return src; + } + + // Handle protocol-relative URLs + if (src.startsWith("//")) { + const protocol = new URL(baseUrl).protocol; + return protocol + src; + } + + // Handle relative URLs + return new URL(src, resolutionBase).href; + } catch (error) { + logger.debug("Failed to resolve image URL", { + src, + baseUrl, + error, + module: "scrapeURL", + method: "extractImages" + }); + return ''; + } +} + +async function extractImagesCheerio(html: string, baseUrl: string): Promise { + const $ = load(html); + const baseHref = $('base[href]').first().attr('href') || ''; + const images: Set = new Set(); + + // Extract from tags + $("img").each((_, element) => { + const src = $(element).attr("src"); + if (src) { + const resolvedUrl = resolveImageUrl(src.trim(), baseUrl, baseHref); + if (resolvedUrl) { + images.add(resolvedUrl); + } + } + + // Also check data-src for lazy-loaded images + const dataSrc = $(element).attr("data-src"); + if (dataSrc) { + const resolvedUrl = resolveImageUrl(dataSrc.trim(), baseUrl, baseHref); + if (resolvedUrl) { + images.add(resolvedUrl); + } + } + + // Check srcset for responsive images + const srcset = $(element).attr("srcset"); + if (srcset) { + // Parse srcset: "url1 1x, url2 2x, ..." + const urls = srcset.split(',').map(s => s.trim().split(/\s+/)[0]); + urls.forEach(url => { + if (url) { + const resolvedUrl = resolveImageUrl(url, baseUrl, baseHref); + if (resolvedUrl) { + images.add(resolvedUrl); + } + } + }); + } + }); + + // Extract from elements + $("picture source").each((_, element) => { + const srcset = $(element).attr("srcset"); + if (srcset) { + const urls = srcset.split(',').map(s => s.trim().split(/\s+/)[0]); + urls.forEach(url => { + if (url) { + const resolvedUrl = resolveImageUrl(url, baseUrl, baseHref); + if (resolvedUrl) { + images.add(resolvedUrl); + } + } + }); + } + }); + + // Extract from meta tags (Open Graph, Twitter Cards) + const metaImages = [ + $('meta[property="og:image"]').attr("content"), + $('meta[property="og:image:url"]').attr("content"), + $('meta[property="og:image:secure_url"]').attr("content"), + $('meta[name="twitter:image"]').attr("content"), + $('meta[name="twitter:image:src"]').attr("content"), + $('meta[itemprop="image"]').attr("content"), + ]; + + metaImages.forEach(src => { + if (src) { + const resolvedUrl = resolveImageUrl(src.trim(), baseUrl, baseHref); + if (resolvedUrl) { + images.add(resolvedUrl); + } + } + }); + + // Extract from link tags (apple-touch-icon, etc.) + $('link[rel*="icon"], link[rel*="apple-touch-icon"], link[rel*="image_src"]').each((_, element) => { + const href = $(element).attr("href"); + if (href) { + const resolvedUrl = resolveImageUrl(href.trim(), baseUrl, baseHref); + if (resolvedUrl) { + images.add(resolvedUrl); + } + } + }); + + // Extract background images from inline styles + $("[style*='background-image']").each((_, element) => { + const style = $(element).attr("style") || ""; + const matches = style.match(/background-image:\s*url\(['"]?([^'")]+)['"]?\)/gi); + if (matches) { + matches.forEach(match => { + const urlMatch = match.match(/url\(['"]?([^'")]+)['"]?\)/i); + if (urlMatch && urlMatch[1]) { + const resolvedUrl = resolveImageUrl(urlMatch[1].trim(), baseUrl, baseHref); + if (resolvedUrl) { + images.add(resolvedUrl); + } + } + }); + } + }); + + // Extract from video poster attributes + $("video[poster]").each((_, element) => { + const poster = $(element).attr("poster"); + if (poster) { + const resolvedUrl = resolveImageUrl(poster.trim(), baseUrl, baseHref); + if (resolvedUrl) { + images.add(resolvedUrl); + } + } + }); + + // Filter out invalid URLs and convert Set to Array + return Array.from(images).filter(url => { + try { + // Skip javascript: URLs for security + if (url.toLowerCase().startsWith('javascript:')) { + return false; + } + new URL(url); + return true; + } catch { + return false; + } + }); +} + +export async function extractImages(html: string, baseUrl: string): Promise { + try { + return await _extractImages(html, baseUrl); + } catch (error) { + logger.warn("Failed to call html-transformer! Falling back to cheerio...", { + error, + module: "scrapeURL", method: "extractImages" + }); + + // Fallback to Cheerio implementation + return await extractImagesCheerio(html, baseUrl); + } +} diff --git a/apps/api/src/scraper/scrapeURL/transformers/index.ts b/apps/api/src/scraper/scrapeURL/transformers/index.ts index f0cde5b22..e3d293d08 100644 --- a/apps/api/src/scraper/scrapeURL/transformers/index.ts +++ b/apps/api/src/scraper/scrapeURL/transformers/index.ts @@ -3,6 +3,7 @@ import { Meta } from ".."; import { Document } from "../../../controllers/v1/types"; import { htmlTransform } from "../lib/removeUnwantedElements"; import { extractLinks } from "../lib/extractLinks"; +import { extractImages } from "../lib/extractImages"; import { extractMetadata } from "../lib/extractMetadata"; import { performLLMExtract, performSummary } from "./llmExtract"; import { uploadScreenshot } from "./uploadScreenshot"; @@ -116,6 +117,21 @@ export async function deriveLinksFromHTML(meta: Meta, document: Document): Promi return document; } +export async function deriveImagesFromHTML(meta: Meta, document: Document): Promise { + // Only derive if the formats has images + if (hasFormatOfType(meta.options.formats, "images")) { + if (document.html === undefined) { + throw new Error( + "html is undefined -- this transformer is being called out of order", + ); + } + + document.images = await extractImages(document.html, document.metadata.url ?? document.metadata.sourceURL ?? meta.rewrittenUrl ?? meta.url); + } + + return document; +} + export function coerceFieldsToFormats( meta: Meta, document: Document, @@ -124,6 +140,7 @@ export function coerceFieldsToFormats( const hasRawHtml = hasFormatOfType(meta.options.formats, "rawHtml"); const hasHtml = hasFormatOfType(meta.options.formats, "html"); const hasLinks = hasFormatOfType(meta.options.formats, "links"); + const hasImages = hasFormatOfType(meta.options.formats, "images"); const hasChangeTracking = hasFormatOfType(meta.options.formats, "changeTracking"); const hasJson = hasFormatOfType(meta.options.formats, "json"); const hasScreenshot = hasFormatOfType(meta.options.formats, "screenshot"); @@ -181,6 +198,19 @@ export function coerceFieldsToFormats( ); } + if (!hasImages && document.images !== undefined) { + meta.logger.warn( + "Removed images from Document because it wasn't in formats -- this is wasteful and indicates a bug.", + { hasImages, hasImagesField: document.images !== undefined } + ); + delete document.images; + } else if (hasImages && document.images === undefined) { + meta.logger.warn( + "Request had format: images, but there was no images field in the result.", + { hasImages, hasImagesField: document.images !== undefined } + ); + } + // Handle v1 backward compatibility - don't delete fields based on v1OriginalFormat const shouldKeepExtract = meta.internalOptions.v1OriginalFormat === "extract"; const shouldKeepJson = meta.internalOptions.v1OriginalFormat === "json"; @@ -269,6 +299,7 @@ export const transformerStack: Transformer[] = [ deriveHTMLFromRawHTML, deriveMarkdownFromHTML, deriveLinksFromHTML, + deriveImagesFromHTML, deriveMetadataFromRawHTML, uploadScreenshot, ...(useIndex ? [sendDocumentToIndex] : []), diff --git a/apps/js-sdk/firecrawl/src/__tests__/e2e/v2/scrape.test.ts b/apps/js-sdk/firecrawl/src/__tests__/e2e/v2/scrape.test.ts index f7b54db65..53c4f12ae 100644 --- a/apps/js-sdk/firecrawl/src/__tests__/e2e/v2/scrape.test.ts +++ b/apps/js-sdk/firecrawl/src/__tests__/e2e/v2/scrape.test.ts @@ -121,6 +121,39 @@ describe("v2.scrape e2e", () => { } }, 90_000); + test("images format: extract all images from webpage", async () => { + if (!client) throw new Error(); + const doc = await client.scrape("https://firecrawl.dev", { + formats: ["images"], + }); + expect(doc.images).toBeTruthy(); + expect(Array.isArray(doc.images)).toBe(true); + expect(doc.images.length).toBeGreaterThan(0); + // Should find firecrawl logo/branding images + expect(doc.images.some(img => img.includes("firecrawl") || img.includes("logo"))).toBe(true); + }, 60_000); + + test("images format: works with multiple formats", async () => { + if (!client) throw new Error(); + const doc = await client.scrape("https://github.com", { + formats: ["markdown", "links", "images"], + }); + expect(doc.markdown).toBeTruthy(); + expect(doc.links).toBeTruthy(); + expect(doc.images).toBeTruthy(); + expect(Array.isArray(doc.images)).toBe(true); + expect(doc.images.length).toBeGreaterThan(0); + + // Images should find things not available in links format + const imageExtensions = ['.jpg', '.jpeg', '.png', '.gif', '.webp', '.svg', '.ico']; + const linkImages = doc.links?.filter(link => + imageExtensions.some(ext => link.toLowerCase().includes(ext)) + ) || []; + + // Should discover additional images beyond those with obvious extensions + expect(doc.images.length).toBeGreaterThanOrEqual(linkImages.length); + }, 60_000); + test("invalid url should throw", async () => { if (!client) throw new Error(); await expect(client.scrape("")).rejects.toThrow("URL cannot be empty"); diff --git a/apps/js-sdk/firecrawl/src/v2/types.ts b/apps/js-sdk/firecrawl/src/v2/types.ts index a52ecc3b4..0e0d76e66 100644 --- a/apps/js-sdk/firecrawl/src/v2/types.ts +++ b/apps/js-sdk/firecrawl/src/v2/types.ts @@ -6,6 +6,7 @@ export type FormatString = | "html" | "rawHtml" | "links" + | "images" | "screenshot" | "summary" | "changeTracking" @@ -167,6 +168,7 @@ export interface Document { summary?: string; metadata?: DocumentMetadata; links?: string[]; + images?: string[]; screenshot?: string; actions?: Record; warning?: string; diff --git a/apps/python-sdk/firecrawl/__tests__/e2e/v2/test_scrape.py b/apps/python-sdk/firecrawl/__tests__/e2e/v2/test_scrape.py index b7d6d8829..57fb5a193 100644 --- a/apps/python-sdk/firecrawl/__tests__/e2e/v2/test_scrape.py +++ b/apps/python-sdk/firecrawl/__tests__/e2e/v2/test_scrape.py @@ -151,4 +151,40 @@ class TestScrapeE2E: max_age=0, store_in_cache=False, ) - assert isinstance(doc, Document) \ No newline at end of file + assert isinstance(doc, Document) + + def test_scrape_images_format(self): + """Test images format extraction.""" + doc = self.client.scrape( + "https://firecrawl.dev", + formats=["images"] + ) + assert isinstance(doc, Document) + assert doc.images is not None + assert isinstance(doc.images, list) + assert len(doc.images) > 0 + # Should find firecrawl logo/branding images + assert any("firecrawl" in img.lower() or "logo" in img.lower() for img in doc.images) + + def test_scrape_images_with_multiple_formats(self): + """Test images format works with other formats.""" + doc = self.client.scrape( + "https://github.com", + formats=["markdown", "links", "images"] + ) + assert isinstance(doc, Document) + assert doc.markdown is not None + assert doc.links is not None + assert doc.images is not None + assert isinstance(doc.images, list) + assert len(doc.images) > 0 + + # Images should find content not available in links format + image_extensions = ['.jpg', '.jpeg', '.png', '.gif', '.webp', '.svg', '.ico'] + link_images = [ + link for link in (doc.links or []) + if any(ext in link.lower() for ext in image_extensions) + ] + + # Should discover additional images beyond those with obvious extensions + assert len(doc.images) >= len(link_images) \ No newline at end of file diff --git a/apps/python-sdk/firecrawl/v2/types.py b/apps/python-sdk/firecrawl/v2/types.py index bff13327c..69e71f830 100644 --- a/apps/python-sdk/firecrawl/v2/types.py +++ b/apps/python-sdk/firecrawl/v2/types.py @@ -123,6 +123,7 @@ class Document(BaseModel): summary: Optional[str] = None metadata: Optional[DocumentMetadata] = None links: Optional[List[str]] = None + images: Optional[List[str]] = None screenshot: Optional[str] = None actions: Optional[Dict[str, Any]] = None warning: Optional[str] = None @@ -182,7 +183,7 @@ CategoryOption = Union[str, Category] FormatString = Literal[ # camelCase versions (API format) - "markdown", "html", "rawHtml", "links", "screenshot", "summary", "changeTracking", "json", + "markdown", "html", "rawHtml", "links", "images", "screenshot", "summary", "changeTracking", "json", # snake_case versions (user-friendly) "raw_html", "change_tracking" ] @@ -226,6 +227,7 @@ class ScrapeFormats(BaseModel): raw_html: bool = False summary: bool = False links: bool = False + images: bool = False screenshot: bool = False change_tracking: bool = False json: bool = False