diff --git a/apps/api/sharedLibs/html-transformer/src/lib.rs b/apps/api/sharedLibs/html-transformer/src/lib.rs
index ff65a4978..a93fa2619 100644
--- a/apps/api/sharedLibs/html-transformer/src/lib.rs
+++ b/apps/api/sharedLibs/html-transformer/src/lib.rs
@@ -533,6 +533,195 @@ pub unsafe extern "C" fn get_inner_json(html: *const libc::c_char) -> *mut libc:
CString::new(out).unwrap().into_raw()
}
+fn _extract_images(html: &str, base_url: &str) -> Result, Box> {
+ let document = parse_html().one(html);
+ let base_url = Url::parse(base_url)?;
+ let base_href = _extract_base_href_from_document(&document, &base_url)?;
+ let base_href_url = Url::parse(&base_href)?;
+ let mut images = HashSet::::new();
+
+ // Helper function to resolve image URLs
+ let resolve_image_url = |src: &str| -> Result> {
+ // Skip data URIs and blob URLs
+ if src.starts_with("data:") || src.starts_with("blob:") {
+ return Ok(src.to_string());
+ }
+
+ // Handle absolute URLs
+ if src.starts_with("http://") || src.starts_with("https://") {
+ return Ok(src.to_string());
+ }
+
+ // Handle protocol-relative URLs
+ if src.starts_with("//") {
+ let resolved = base_url.join(src)?;
+ return Ok(resolved.to_string());
+ }
+
+ // Handle relative URLs
+ let resolved = base_href_url.join(src)?;
+ Ok(resolved.to_string())
+ };
+
+ // Extract from img tags
+ let img_elements: Vec<_> = match document.select("img").map_err(|_| "Failed to select img tags") {
+ Ok(x) => x.collect(),
+ Err(e) => return Err(e.into()),
+ };
+
+ for img in img_elements {
+ let attrs = img.attributes.borrow();
+
+ // Extract from src attribute
+ if let Some(src) = attrs.get("src") {
+ if let Ok(resolved) = resolve_image_url(src) {
+ images.insert(resolved);
+ }
+ }
+
+ // Extract from data-src (lazy loading)
+ if let Some(data_src) = attrs.get("data-src") {
+ if let Ok(resolved) = resolve_image_url(data_src) {
+ images.insert(resolved);
+ }
+ }
+
+ // Extract from srcset (responsive images)
+ if let Some(srcset) = attrs.get("srcset") {
+ for part in srcset.split(',') {
+ if let Some(url) = part.trim().split_whitespace().next() {
+ if !url.is_empty() {
+ if let Ok(resolved) = resolve_image_url(url) {
+ images.insert(resolved);
+ }
+ }
+ }
+ }
+ }
+ }
+
+ // Extract from picture source elements
+ let source_elements: Vec<_> = match document.select("picture source").map_err(|_| "Failed to select picture source") {
+ Ok(x) => x.collect(),
+ Err(_) => Vec::new(),
+ };
+
+ for source in source_elements {
+ if let Some(srcset) = source.attributes.borrow().get("srcset") {
+ for part in srcset.split(',') {
+ if let Some(url) = part.trim().split_whitespace().next() {
+ if !url.is_empty() {
+ if let Ok(resolved) = resolve_image_url(url) {
+ images.insert(resolved);
+ }
+ }
+ }
+ }
+ }
+ }
+
+ // Extract from meta tags (Open Graph, Twitter Cards)
+ let meta_selectors = [
+ "meta[property=\"og:image\"]",
+ "meta[property=\"og:image:url\"]",
+ "meta[property=\"og:image:secure_url\"]",
+ "meta[name=\"twitter:image\"]",
+ "meta[name=\"twitter:image:src\"]",
+ "meta[itemprop=\"image\"]"
+ ];
+
+ for selector in &meta_selectors {
+ if let Ok(elements) = document.select(selector) {
+ for element in elements {
+ if let Some(content) = element.attributes.borrow().get("content") {
+ if let Ok(resolved) = resolve_image_url(content) {
+ images.insert(resolved);
+ }
+ }
+ }
+ }
+ }
+
+ // Extract from link tags (favicons, apple-touch-icons)
+ let link_selectors = [
+ "link[rel*=\"icon\"]",
+ "link[rel*=\"apple-touch-icon\"]",
+ "link[rel*=\"image_src\"]"
+ ];
+
+ for selector in &link_selectors {
+ if let Ok(elements) = document.select(selector) {
+ for element in elements {
+ if let Some(href) = element.attributes.borrow().get("href") {
+ if let Ok(resolved) = resolve_image_url(href) {
+ images.insert(resolved);
+ }
+ }
+ }
+ }
+ }
+
+ // Extract from video poster attributes
+ if let Ok(video_elements) = document.select("video[poster]") {
+ for video in video_elements {
+ if let Some(poster) = video.attributes.borrow().get("poster") {
+ if let Ok(resolved) = resolve_image_url(poster) {
+ images.insert(resolved);
+ }
+ }
+ }
+ }
+
+ // Filter out javascript: URLs for security and validate URLs
+ let filtered_images: Vec = images.into_iter()
+ .filter(|url| !url.to_lowercase().starts_with("javascript:"))
+ .filter(|url| !url.is_empty())
+ .filter(|url| {
+ // Validate URLs (allow data URIs and valid URLs)
+ url.starts_with("data:") || url.starts_with("blob:") || Url::parse(url).is_ok()
+ })
+ .collect();
+
+ Ok(filtered_images)
+}
+
+/// Extracts images from HTML
+///
+/// # Safety
+/// Input must be a C HTML string and base URL string. Output will be a JSON string array. Output string must be freed with free_string.
+#[no_mangle]
+pub unsafe extern "C" fn extract_images(html: *const libc::c_char, base_url: *const libc::c_char) -> *mut libc::c_char {
+ let html = match unsafe { CStr::from_ptr(html) }.to_str().map_err(|_| ()) {
+ Ok(x) => x,
+ Err(_) => {
+ return CString::new("RUSTFC:ERROR:Failed to parse input HTML as C string").unwrap().into_raw();
+ }
+ };
+
+ let base_url = match unsafe { CStr::from_ptr(base_url) }.to_str().map_err(|_| ()) {
+ Ok(x) => x,
+ Err(_) => {
+ return CString::new("RUSTFC:ERROR:Failed to parse input base URL as C string").unwrap().into_raw();
+ }
+ };
+
+ let images = match _extract_images(html, base_url) {
+ Ok(x) => x,
+ Err(e) => {
+ return CString::new(format!("RUSTFC:ERROR:{}", e)).unwrap().into_raw();
+ }
+ };
+
+ let images_out = match serde_json::ser::to_string(&images) {
+ Ok(x) => x,
+ Err(e) => {
+ return CString::new(format!("RUSTFC:ERROR:{}", e)).unwrap().into_raw();
+ }
+ };
+
+ CString::new(images_out).unwrap().into_raw()
+}
+
/// Frees a string allocated in Rust-land.
///
/// # Safety
diff --git a/apps/api/src/__tests__/snips/v2/scrape.test.ts b/apps/api/src/__tests__/snips/v2/scrape.test.ts
index 55761395b..da70a777b 100644
--- a/apps/api/src/__tests__/snips/v2/scrape.test.ts
+++ b/apps/api/src/__tests__/snips/v2/scrape.test.ts
@@ -120,6 +120,39 @@ describe("Scrape tests", () => {
expect(response.links?.length).toBeGreaterThan(0);
});
+ it.concurrent("images format works", async () => {
+ const response = await scrape({
+ url: "https://firecrawl.dev",
+ formats: ["images"],
+ }, identity);
+
+ expect(response.images).toBeDefined();
+ expect(response.images?.length).toBeGreaterThan(0);
+ // Firecrawl website should have at least the logo
+ expect(response.images?.some(img => img.includes("firecrawl"))).toBe(true);
+ });
+
+ it.concurrent("images format works with multiple formats", async () => {
+ const response = await scrape({
+ url: "https://firecrawl.dev",
+ formats: ["markdown", "links", "images"],
+ }, identity);
+
+ expect(response.markdown).toBeDefined();
+ expect(response.links).toBeDefined();
+ expect(response.images).toBeDefined();
+ expect(response.images?.length).toBeGreaterThan(0);
+
+ // Images should include things that aren't in links
+ const imageExtensions = ['.jpg', '.jpeg', '.png', '.gif', '.webp', '.svg', '.ico'];
+ const linkImages = response.links?.filter(link =>
+ imageExtensions.some(ext => link.toLowerCase().includes(ext))
+ ) || [];
+
+ // Should have found more images than just those with obvious extensions in links
+ expect(response.images?.length).toBeGreaterThanOrEqual(linkImages.length);
+ });
+
if (process.env.TEST_SUITE_SELF_HOSTED && process.env.PROXY_SERVER) {
it.concurrent("self-hosted proxy works", async () => {
const response = await scrape({
diff --git a/apps/api/src/controllers/v1/types.ts b/apps/api/src/controllers/v1/types.ts
index 9cee77a61..2e0055888 100644
--- a/apps/api/src/controllers/v1/types.ts
+++ b/apps/api/src/controllers/v1/types.ts
@@ -829,6 +829,7 @@ export type Document = {
html?: string;
rawHtml?: string;
links?: string[];
+ images?: string[];
screenshot?: string;
extract?: any;
json?: any;
diff --git a/apps/api/src/controllers/v2/types.ts b/apps/api/src/controllers/v2/types.ts
index fa2c8b225..d7222bf01 100644
--- a/apps/api/src/controllers/v2/types.ts
+++ b/apps/api/src/controllers/v2/types.ts
@@ -35,6 +35,7 @@ export type Format =
| "html"
| "rawHtml"
| "links"
+ | "images"
| "screenshot"
| "screenshot@fullPage"
| "extract"
@@ -216,6 +217,7 @@ export type FormatObject =
| { type: "html" }
| { type: "rawHtml" }
| { type: "links" }
+ | { type: "images" }
| { type: "summary" }
| JsonFormatWithOptions
| ChangeTrackingFormatWithOptions
@@ -250,6 +252,7 @@ const baseScrapeOptions = z
z.object({ type: z.literal("html") }),
z.object({ type: z.literal("rawHtml") }),
z.object({ type: z.literal("links") }),
+ z.object({ type: z.literal("images") }),
z.object({ type: z.literal("summary") }),
jsonFormatWithOptions,
changeTrackingFormatWithOptions,
@@ -629,6 +632,7 @@ export type Document = {
html?: string;
rawHtml?: string;
links?: string[];
+ images?: string[];
screenshot?: string;
extract?: any;
json?: any;
@@ -1349,6 +1353,7 @@ export const searchRequestSchema = z
z.object({ type: z.literal("html") }),
z.object({ type: z.literal("rawHtml") }),
z.object({ type: z.literal("links") }),
+ z.object({ type: z.literal("images") }),
z.object({ type: z.literal("summary") }),
jsonFormatWithOptions,
screenshotFormatWithOptions,
diff --git a/apps/api/src/lib/html-transformer.ts b/apps/api/src/lib/html-transformer.ts
index ce334292f..68bf6eb4e 100644
--- a/apps/api/src/lib/html-transformer.ts
+++ b/apps/api/src/lib/html-transformer.ts
@@ -22,6 +22,7 @@ type TransformHtmlOptions = {
class RustHTMLTransformer {
private static instance: RustHTMLTransformer;
private _extractLinks: KoffiFunction;
+ private _extractImages: KoffiFunction;
private _extractBaseHref: KoffiFunction;
private _extractMetadata: KoffiFunction;
private _transformHtml: KoffiFunction;
@@ -34,6 +35,7 @@ class RustHTMLTransformer {
const cstn = "CString:" + crypto.randomUUID();
const freedResultString = koffi.disposable(cstn, "string", this._freeString);
this._extractLinks = lib.func("extract_links", freedResultString, ["string"]);
+ this._extractImages = lib.func("extract_images", freedResultString, ["string", "string"]);
this._extractBaseHref = lib.func("extract_base_href", freedResultString, ["string", "string"]);
this._extractMetadata = lib.func("extract_metadata", freedResultString, ["string"]);
this._transformHtml = lib.func("transform_html", freedResultString, ["string"]);
@@ -64,6 +66,22 @@ class RustHTMLTransformer {
});
}
+ public async extractImages(html: string, baseUrl: string): Promise {
+ return new Promise((resolve, reject) => {
+ this._extractImages.async(html, baseUrl, (err: Error, res: string) => {
+ if (err) {
+ reject(err);
+ } else {
+ if (res.startsWith("RUSTFC:ERROR:")) {
+ reject(new Error(res.replace("RUSTFC:ERROR:", "")));
+ } else {
+ resolve(JSON.parse(res));
+ }
+ }
+ });
+ });
+ }
+
public async extractBaseHref(html: string, url: string): Promise {
return new Promise((resolve, reject) => {
this._extractBaseHref.async(html, url, (err: Error, res: string) => {
@@ -132,6 +150,18 @@ export async function extractLinks(
return await converter.extractLinks(html);
}
+export async function extractImages(
+ html: string | null | undefined,
+ baseUrl: string = ''
+): Promise {
+ if (!html) {
+ return [];
+ }
+
+ const converter = await RustHTMLTransformer.getInstance();
+ return await converter.extractImages(html, baseUrl);
+}
+
export async function extractBaseHref(
html: string | null | undefined,
url: string
diff --git a/apps/api/src/scraper/scrapeURL/lib/__tests__/extractImages.test.ts b/apps/api/src/scraper/scrapeURL/lib/__tests__/extractImages.test.ts
new file mode 100644
index 000000000..b8ad9887f
--- /dev/null
+++ b/apps/api/src/scraper/scrapeURL/lib/__tests__/extractImages.test.ts
@@ -0,0 +1,231 @@
+import { extractImages } from '../extractImages';
+
+describe('extractImages', () => {
+ const baseUrl = 'https://example.com/page.html';
+
+ it('should extract images from img tags', async () => {
+ const html = `
+
+
+
+
+
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ expect(images).toContain('https://example.com/image1.jpg');
+ expect(images).toContain('https://example.com/images/image2.png');
+ expect(images).toContain('https://external.com/image3.gif');
+ expect(images).toHaveLength(3);
+ });
+
+ it('should extract lazy-loaded images from data-src', async () => {
+ const html = `
+
+
+
+
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ expect(images).toContain('https://example.com/lazy-image.webp');
+ expect(images).toContain('https://example.com/regular.jpg');
+ expect(images).toContain('https://example.com/ignored.jpg');
+ });
+
+ it('should extract images from srcset', async () => {
+ const html = `
+
+
+
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ expect(images).toContain('https://example.com/small.jpg');
+ expect(images).toContain('https://example.com/medium.jpg');
+ expect(images).toContain('https://example.com/large.jpg');
+ expect(images).toHaveLength(3);
+ });
+
+ it('should extract images from picture elements', async () => {
+ const html = `
+
+
+
+
+
+
+
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ expect(images).toContain('https://example.com/image.avif');
+ expect(images).toContain('https://example.com/image.webp');
+ expect(images).toContain('https://example.com/image.jpg');
+ });
+
+ it('should extract images from meta tags', async () => {
+ const html = `
+
+
+
+
+
+
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ expect(images).toContain('https://example.com/og-image.jpg');
+ expect(images).toContain('https://example.com/og-image-secure.jpg');
+ expect(images).toContain('https://example.com/twitter-image.png');
+ expect(images).toContain('https://example.com/schema-image.jpg');
+ });
+
+ it('should extract images from link tags', async () => {
+ const html = `
+
+
+
+
+
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ expect(images).toContain('https://example.com/favicon.ico');
+ expect(images).toContain('https://example.com/apple-touch-icon.png');
+ expect(images).toContain('https://example.com/link-image.jpg');
+ });
+
+ it('should extract background images from inline styles', async () => {
+ const html = `
+
+
+ Content
+ Content
+ Content
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ expect(images).toContain('https://example.com/background1.jpg');
+ expect(images).toContain('https://example.com/images/background2.png');
+ expect(images).toContain('https://example.com/background3.gif');
+ });
+
+ it('should extract video poster images', async () => {
+ const html = `
+
+
+
+
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ expect(images).toContain('https://example.com/video-poster.jpg');
+ expect(images).toContain('https://example.com/videos/poster.png');
+ });
+
+ it('should handle protocol-relative URLs', async () => {
+ const html = `
+
+
+
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ expect(images).toContain('https://cdn.example.com/image.jpg');
+ });
+
+ it('should handle data URIs', async () => {
+ const html = `
+
+
+
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ expect(images).toContain('data:image/png;base64,iVBORw0KGgoAAAANS...');
+ });
+
+ it('should respect base tag', async () => {
+ const html = `
+
+
+
+
+
+
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ expect(images).toContain('https://different.com/base/image.jpg');
+ });
+
+ it('should remove duplicates', async () => {
+ const html = `
+
+
+
+
+
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ const duplicateCount = images.filter(img => img.includes('duplicate.jpg')).length;
+ expect(duplicateCount).toBe(1);
+ });
+
+ it('should handle invalid URLs gracefully', async () => {
+ const html = `
+
+
+
+
+
+
+
+
+ `;
+
+ const images = await extractImages(html, baseUrl);
+
+ expect(images).toContain('https://example.com/valid.jpg');
+ expect(images).not.toContain("javascript:alert('xss')");
+ expect(images).toHaveLength(1);
+ });
+});
diff --git a/apps/api/src/scraper/scrapeURL/lib/extractImages.ts b/apps/api/src/scraper/scrapeURL/lib/extractImages.ts
new file mode 100644
index 000000000..80f509256
--- /dev/null
+++ b/apps/api/src/scraper/scrapeURL/lib/extractImages.ts
@@ -0,0 +1,193 @@
+import { load } from "cheerio";
+import { logger } from "../../../lib/logger";
+import { extractImages as _extractImages } from "../../../lib/html-transformer";
+
+function resolveImageUrl(src: string, baseUrl: string, baseHref: string = ''): string {
+ let resolutionBase = baseUrl;
+
+ if (baseHref) {
+ try {
+ new URL(baseHref);
+ resolutionBase = baseHref;
+ } catch {
+ try {
+ resolutionBase = new URL(baseHref, baseUrl).href;
+ } catch {
+ resolutionBase = baseUrl;
+ }
+ }
+ }
+
+ try {
+ // Skip data URIs and blob URLs
+ if (src.startsWith("data:") || src.startsWith("blob:")) {
+ return src;
+ }
+
+ // Handle absolute URLs
+ if (src.startsWith("http://") || src.startsWith("https://")) {
+ return src;
+ }
+
+ // Handle protocol-relative URLs
+ if (src.startsWith("//")) {
+ const protocol = new URL(baseUrl).protocol;
+ return protocol + src;
+ }
+
+ // Handle relative URLs
+ return new URL(src, resolutionBase).href;
+ } catch (error) {
+ logger.debug("Failed to resolve image URL", {
+ src,
+ baseUrl,
+ error,
+ module: "scrapeURL",
+ method: "extractImages"
+ });
+ return '';
+ }
+}
+
+async function extractImagesCheerio(html: string, baseUrl: string): Promise {
+ const $ = load(html);
+ const baseHref = $('base[href]').first().attr('href') || '';
+ const images: Set = new Set();
+
+ // Extract from
tags
+ $("img").each((_, element) => {
+ const src = $(element).attr("src");
+ if (src) {
+ const resolvedUrl = resolveImageUrl(src.trim(), baseUrl, baseHref);
+ if (resolvedUrl) {
+ images.add(resolvedUrl);
+ }
+ }
+
+ // Also check data-src for lazy-loaded images
+ const dataSrc = $(element).attr("data-src");
+ if (dataSrc) {
+ const resolvedUrl = resolveImageUrl(dataSrc.trim(), baseUrl, baseHref);
+ if (resolvedUrl) {
+ images.add(resolvedUrl);
+ }
+ }
+
+ // Check srcset for responsive images
+ const srcset = $(element).attr("srcset");
+ if (srcset) {
+ // Parse srcset: "url1 1x, url2 2x, ..."
+ const urls = srcset.split(',').map(s => s.trim().split(/\s+/)[0]);
+ urls.forEach(url => {
+ if (url) {
+ const resolvedUrl = resolveImageUrl(url, baseUrl, baseHref);
+ if (resolvedUrl) {
+ images.add(resolvedUrl);
+ }
+ }
+ });
+ }
+ });
+
+ // Extract from elements
+ $("picture source").each((_, element) => {
+ const srcset = $(element).attr("srcset");
+ if (srcset) {
+ const urls = srcset.split(',').map(s => s.trim().split(/\s+/)[0]);
+ urls.forEach(url => {
+ if (url) {
+ const resolvedUrl = resolveImageUrl(url, baseUrl, baseHref);
+ if (resolvedUrl) {
+ images.add(resolvedUrl);
+ }
+ }
+ });
+ }
+ });
+
+ // Extract from meta tags (Open Graph, Twitter Cards)
+ const metaImages = [
+ $('meta[property="og:image"]').attr("content"),
+ $('meta[property="og:image:url"]').attr("content"),
+ $('meta[property="og:image:secure_url"]').attr("content"),
+ $('meta[name="twitter:image"]').attr("content"),
+ $('meta[name="twitter:image:src"]').attr("content"),
+ $('meta[itemprop="image"]').attr("content"),
+ ];
+
+ metaImages.forEach(src => {
+ if (src) {
+ const resolvedUrl = resolveImageUrl(src.trim(), baseUrl, baseHref);
+ if (resolvedUrl) {
+ images.add(resolvedUrl);
+ }
+ }
+ });
+
+ // Extract from link tags (apple-touch-icon, etc.)
+ $('link[rel*="icon"], link[rel*="apple-touch-icon"], link[rel*="image_src"]').each((_, element) => {
+ const href = $(element).attr("href");
+ if (href) {
+ const resolvedUrl = resolveImageUrl(href.trim(), baseUrl, baseHref);
+ if (resolvedUrl) {
+ images.add(resolvedUrl);
+ }
+ }
+ });
+
+ // Extract background images from inline styles
+ $("[style*='background-image']").each((_, element) => {
+ const style = $(element).attr("style") || "";
+ const matches = style.match(/background-image:\s*url\(['"]?([^'")]+)['"]?\)/gi);
+ if (matches) {
+ matches.forEach(match => {
+ const urlMatch = match.match(/url\(['"]?([^'")]+)['"]?\)/i);
+ if (urlMatch && urlMatch[1]) {
+ const resolvedUrl = resolveImageUrl(urlMatch[1].trim(), baseUrl, baseHref);
+ if (resolvedUrl) {
+ images.add(resolvedUrl);
+ }
+ }
+ });
+ }
+ });
+
+ // Extract from video poster attributes
+ $("video[poster]").each((_, element) => {
+ const poster = $(element).attr("poster");
+ if (poster) {
+ const resolvedUrl = resolveImageUrl(poster.trim(), baseUrl, baseHref);
+ if (resolvedUrl) {
+ images.add(resolvedUrl);
+ }
+ }
+ });
+
+ // Filter out invalid URLs and convert Set to Array
+ return Array.from(images).filter(url => {
+ try {
+ // Skip javascript: URLs for security
+ if (url.toLowerCase().startsWith('javascript:')) {
+ return false;
+ }
+ new URL(url);
+ return true;
+ } catch {
+ return false;
+ }
+ });
+}
+
+export async function extractImages(html: string, baseUrl: string): Promise {
+ try {
+ return await _extractImages(html, baseUrl);
+ } catch (error) {
+ logger.warn("Failed to call html-transformer! Falling back to cheerio...", {
+ error,
+ module: "scrapeURL", method: "extractImages"
+ });
+
+ // Fallback to Cheerio implementation
+ return await extractImagesCheerio(html, baseUrl);
+ }
+}
diff --git a/apps/api/src/scraper/scrapeURL/transformers/index.ts b/apps/api/src/scraper/scrapeURL/transformers/index.ts
index f0cde5b22..e3d293d08 100644
--- a/apps/api/src/scraper/scrapeURL/transformers/index.ts
+++ b/apps/api/src/scraper/scrapeURL/transformers/index.ts
@@ -3,6 +3,7 @@ import { Meta } from "..";
import { Document } from "../../../controllers/v1/types";
import { htmlTransform } from "../lib/removeUnwantedElements";
import { extractLinks } from "../lib/extractLinks";
+import { extractImages } from "../lib/extractImages";
import { extractMetadata } from "../lib/extractMetadata";
import { performLLMExtract, performSummary } from "./llmExtract";
import { uploadScreenshot } from "./uploadScreenshot";
@@ -116,6 +117,21 @@ export async function deriveLinksFromHTML(meta: Meta, document: Document): Promi
return document;
}
+export async function deriveImagesFromHTML(meta: Meta, document: Document): Promise {
+ // Only derive if the formats has images
+ if (hasFormatOfType(meta.options.formats, "images")) {
+ if (document.html === undefined) {
+ throw new Error(
+ "html is undefined -- this transformer is being called out of order",
+ );
+ }
+
+ document.images = await extractImages(document.html, document.metadata.url ?? document.metadata.sourceURL ?? meta.rewrittenUrl ?? meta.url);
+ }
+
+ return document;
+}
+
export function coerceFieldsToFormats(
meta: Meta,
document: Document,
@@ -124,6 +140,7 @@ export function coerceFieldsToFormats(
const hasRawHtml = hasFormatOfType(meta.options.formats, "rawHtml");
const hasHtml = hasFormatOfType(meta.options.formats, "html");
const hasLinks = hasFormatOfType(meta.options.formats, "links");
+ const hasImages = hasFormatOfType(meta.options.formats, "images");
const hasChangeTracking = hasFormatOfType(meta.options.formats, "changeTracking");
const hasJson = hasFormatOfType(meta.options.formats, "json");
const hasScreenshot = hasFormatOfType(meta.options.formats, "screenshot");
@@ -181,6 +198,19 @@ export function coerceFieldsToFormats(
);
}
+ if (!hasImages && document.images !== undefined) {
+ meta.logger.warn(
+ "Removed images from Document because it wasn't in formats -- this is wasteful and indicates a bug.",
+ { hasImages, hasImagesField: document.images !== undefined }
+ );
+ delete document.images;
+ } else if (hasImages && document.images === undefined) {
+ meta.logger.warn(
+ "Request had format: images, but there was no images field in the result.",
+ { hasImages, hasImagesField: document.images !== undefined }
+ );
+ }
+
// Handle v1 backward compatibility - don't delete fields based on v1OriginalFormat
const shouldKeepExtract = meta.internalOptions.v1OriginalFormat === "extract";
const shouldKeepJson = meta.internalOptions.v1OriginalFormat === "json";
@@ -269,6 +299,7 @@ export const transformerStack: Transformer[] = [
deriveHTMLFromRawHTML,
deriveMarkdownFromHTML,
deriveLinksFromHTML,
+ deriveImagesFromHTML,
deriveMetadataFromRawHTML,
uploadScreenshot,
...(useIndex ? [sendDocumentToIndex] : []),
diff --git a/apps/js-sdk/firecrawl/src/__tests__/e2e/v2/scrape.test.ts b/apps/js-sdk/firecrawl/src/__tests__/e2e/v2/scrape.test.ts
index f7b54db65..53c4f12ae 100644
--- a/apps/js-sdk/firecrawl/src/__tests__/e2e/v2/scrape.test.ts
+++ b/apps/js-sdk/firecrawl/src/__tests__/e2e/v2/scrape.test.ts
@@ -121,6 +121,39 @@ describe("v2.scrape e2e", () => {
}
}, 90_000);
+ test("images format: extract all images from webpage", async () => {
+ if (!client) throw new Error();
+ const doc = await client.scrape("https://firecrawl.dev", {
+ formats: ["images"],
+ });
+ expect(doc.images).toBeTruthy();
+ expect(Array.isArray(doc.images)).toBe(true);
+ expect(doc.images.length).toBeGreaterThan(0);
+ // Should find firecrawl logo/branding images
+ expect(doc.images.some(img => img.includes("firecrawl") || img.includes("logo"))).toBe(true);
+ }, 60_000);
+
+ test("images format: works with multiple formats", async () => {
+ if (!client) throw new Error();
+ const doc = await client.scrape("https://github.com", {
+ formats: ["markdown", "links", "images"],
+ });
+ expect(doc.markdown).toBeTruthy();
+ expect(doc.links).toBeTruthy();
+ expect(doc.images).toBeTruthy();
+ expect(Array.isArray(doc.images)).toBe(true);
+ expect(doc.images.length).toBeGreaterThan(0);
+
+ // Images should find things not available in links format
+ const imageExtensions = ['.jpg', '.jpeg', '.png', '.gif', '.webp', '.svg', '.ico'];
+ const linkImages = doc.links?.filter(link =>
+ imageExtensions.some(ext => link.toLowerCase().includes(ext))
+ ) || [];
+
+ // Should discover additional images beyond those with obvious extensions
+ expect(doc.images.length).toBeGreaterThanOrEqual(linkImages.length);
+ }, 60_000);
+
test("invalid url should throw", async () => {
if (!client) throw new Error();
await expect(client.scrape("")).rejects.toThrow("URL cannot be empty");
diff --git a/apps/js-sdk/firecrawl/src/v2/types.ts b/apps/js-sdk/firecrawl/src/v2/types.ts
index a52ecc3b4..0e0d76e66 100644
--- a/apps/js-sdk/firecrawl/src/v2/types.ts
+++ b/apps/js-sdk/firecrawl/src/v2/types.ts
@@ -6,6 +6,7 @@ export type FormatString =
| "html"
| "rawHtml"
| "links"
+ | "images"
| "screenshot"
| "summary"
| "changeTracking"
@@ -167,6 +168,7 @@ export interface Document {
summary?: string;
metadata?: DocumentMetadata;
links?: string[];
+ images?: string[];
screenshot?: string;
actions?: Record;
warning?: string;
diff --git a/apps/python-sdk/firecrawl/__tests__/e2e/v2/test_scrape.py b/apps/python-sdk/firecrawl/__tests__/e2e/v2/test_scrape.py
index b7d6d8829..57fb5a193 100644
--- a/apps/python-sdk/firecrawl/__tests__/e2e/v2/test_scrape.py
+++ b/apps/python-sdk/firecrawl/__tests__/e2e/v2/test_scrape.py
@@ -151,4 +151,40 @@ class TestScrapeE2E:
max_age=0,
store_in_cache=False,
)
- assert isinstance(doc, Document)
\ No newline at end of file
+ assert isinstance(doc, Document)
+
+ def test_scrape_images_format(self):
+ """Test images format extraction."""
+ doc = self.client.scrape(
+ "https://firecrawl.dev",
+ formats=["images"]
+ )
+ assert isinstance(doc, Document)
+ assert doc.images is not None
+ assert isinstance(doc.images, list)
+ assert len(doc.images) > 0
+ # Should find firecrawl logo/branding images
+ assert any("firecrawl" in img.lower() or "logo" in img.lower() for img in doc.images)
+
+ def test_scrape_images_with_multiple_formats(self):
+ """Test images format works with other formats."""
+ doc = self.client.scrape(
+ "https://github.com",
+ formats=["markdown", "links", "images"]
+ )
+ assert isinstance(doc, Document)
+ assert doc.markdown is not None
+ assert doc.links is not None
+ assert doc.images is not None
+ assert isinstance(doc.images, list)
+ assert len(doc.images) > 0
+
+ # Images should find content not available in links format
+ image_extensions = ['.jpg', '.jpeg', '.png', '.gif', '.webp', '.svg', '.ico']
+ link_images = [
+ link for link in (doc.links or [])
+ if any(ext in link.lower() for ext in image_extensions)
+ ]
+
+ # Should discover additional images beyond those with obvious extensions
+ assert len(doc.images) >= len(link_images)
\ No newline at end of file
diff --git a/apps/python-sdk/firecrawl/v2/types.py b/apps/python-sdk/firecrawl/v2/types.py
index bff13327c..69e71f830 100644
--- a/apps/python-sdk/firecrawl/v2/types.py
+++ b/apps/python-sdk/firecrawl/v2/types.py
@@ -123,6 +123,7 @@ class Document(BaseModel):
summary: Optional[str] = None
metadata: Optional[DocumentMetadata] = None
links: Optional[List[str]] = None
+ images: Optional[List[str]] = None
screenshot: Optional[str] = None
actions: Optional[Dict[str, Any]] = None
warning: Optional[str] = None
@@ -182,7 +183,7 @@ CategoryOption = Union[str, Category]
FormatString = Literal[
# camelCase versions (API format)
- "markdown", "html", "rawHtml", "links", "screenshot", "summary", "changeTracking", "json",
+ "markdown", "html", "rawHtml", "links", "images", "screenshot", "summary", "changeTracking", "json",
# snake_case versions (user-friendly)
"raw_html", "change_tracking"
]
@@ -226,6 +227,7 @@ class ScrapeFormats(BaseModel):
raw_html: bool = False
summary: bool = False
links: bool = False
+ images: bool = False
screenshot: bool = False
change_tracking: bool = False
json: bool = False