diff --git a/apps/api/sharedLibs/html-transformer/src/lib.rs b/apps/api/sharedLibs/html-transformer/src/lib.rs
index a93fa2619..7f791b38e 100644
--- a/apps/api/sharedLibs/html-transformer/src/lib.rs
+++ b/apps/api/sharedLibs/html-transformer/src/lib.rs
@@ -533,6 +533,90 @@ pub unsafe extern "C" fn get_inner_json(html: *const libc::c_char) -> *mut libc:
CString::new(out).unwrap().into_raw()
}
+#[derive(Deserialize)]
+struct AttributeSelector {
+ selector: String,
+ attribute: String,
+}
+
+#[derive(Deserialize)]
+struct ExtractAttributesOptions {
+ selectors: Vec,
+}
+
+fn _extract_attributes(html: &str, options: &ExtractAttributesOptions) -> Result> {
+ let document = parse_html().one(html);
+ let mut results = Vec::new();
+
+ for selector_config in &options.selectors {
+ let mut values = Vec::new();
+
+ let elements: Vec<_> = match document.select(&selector_config.selector).map_err(|_| format!("Failed to select with selector: {}", selector_config.selector)) {
+ Ok(x) => x.collect(),
+ Err(_) => {
+ // Return empty values for invalid selectors rather than failing completely
+ Vec::new()
+ }
+ };
+
+ for element in elements {
+ // Try the attribute as-is first
+ if let Some(attr_value) = element.attributes.borrow().get(selector_config.attribute.as_str()) {
+ values.push(attr_value.to_string());
+ continue;
+ }
+
+ // If not found and doesn't start with 'data-', try with 'data-' prefix
+ if !selector_config.attribute.starts_with("data-") {
+ let data_attr = format!("data-{}", selector_config.attribute);
+ if let Some(attr_value) = element.attributes.borrow().get(data_attr.as_str()) {
+ values.push(attr_value.to_string());
+ }
+ }
+ }
+
+ results.push(serde_json::json!({
+ "selector": selector_config.selector,
+ "attribute": selector_config.attribute,
+ "values": values
+ }));
+ }
+
+ Ok(serde_json::to_string(&results)?)
+}
+
+/// Extracts attributes from HTML elements using CSS selectors
+///
+/// # Safety
+/// html must be a C HTML string. options must be a C JSON string with format:
+/// {"selectors": [{"selector": "...", "attribute": "..."}]}
+/// Output will be a JSON string. Output string must be freed with free_string.
+#[no_mangle]
+pub unsafe extern "C" fn extract_attributes(html: *const libc::c_char, options: *const libc::c_char) -> *mut libc::c_char {
+ let options_str = match unsafe { CStr::from_ptr(options) }.to_str().map_err(|_| ()) {
+ Ok(x) => x,
+ Err(_) => {
+ return CString::new("RUSTFC:ERROR:Failed to parse input options as C string").unwrap().into_raw();
+ }
+ };
+
+ let parsed_options: ExtractAttributesOptions = match serde_json::from_str(options_str) {
+ Ok(x) => x,
+ Err(e) => {
+ return CString::new(format!("RUSTFC:ERROR:Failed to parse options JSON: {}", e)).unwrap().into_raw();
+ }
+ };
+
+ let result = match _extract_attributes(html, &parsed_options) {
+ Ok(x) => x,
+ Err(e) => {
+ return CString::new(format!("RUSTFC:ERROR:{}", e)).unwrap().into_raw();
+ }
+ };
+
+ CString::new(result).unwrap().into_raw()
+}
+
fn _extract_images(html: &str, base_url: &str) -> Result, Box> {
let document = parse_html().one(html);
let base_url = Url::parse(base_url)?;
diff --git a/apps/api/src/__tests__/snips/v2/scrape.test.ts b/apps/api/src/__tests__/snips/v2/scrape.test.ts
index da70a777b..315f93cca 100644
--- a/apps/api/src/__tests__/snips/v2/scrape.test.ts
+++ b/apps/api/src/__tests__/snips/v2/scrape.test.ts
@@ -1001,3 +1001,73 @@ describe("Scrape tests", () => {
}, scrapeTimeout);
});
});
+
+describe("attributes format", () => {
+ it.concurrent("should extract attributes from HTML elements", async () => {
+ const response = await scrape({
+ url: "https://news.ycombinator.com",
+ formats: [
+ { type: "markdown" },
+ {
+ type: "attributes",
+ selectors: [
+ { selector: ".athing", attribute: "id" }
+ ]
+ }
+ ]
+ }, identity);
+
+ expect(response.markdown).toBeDefined();
+ expect(response.attributes).toBeDefined();
+ expect(Array.isArray(response.attributes)).toBe(true);
+ expect(response.attributes!.length).toBe(1);
+ expect(response.attributes![0]).toEqual({
+ selector: ".athing",
+ attribute: "id",
+ values: expect.any(Array)
+ });
+ expect(response.attributes![0].values.length).toBeGreaterThan(0);
+ }, scrapeTimeout);
+
+ it.concurrent("should handle multiple attribute selectors", async () => {
+ const response = await scrape({
+ url: "https://github.com/microsoft/vscode",
+ formats: [
+ {
+ type: "attributes",
+ selectors: [
+ { selector: "[data-testid]", attribute: "data-testid" },
+ { selector: "[data-view-component]", attribute: "data-view-component" }
+ ]
+ }
+ ]
+ }, identity);
+
+ expect(response.attributes).toBeDefined();
+ expect(Array.isArray(response.attributes)).toBe(true);
+ expect(response.attributes!.length).toBe(2);
+
+ const testIdResults = response.attributes!.find(a => a.attribute === "data-testid");
+ expect(testIdResults).toBeDefined();
+ expect(testIdResults!.selector).toBe("[data-testid]");
+ }, scrapeTimeout);
+
+ it.concurrent("should return empty arrays when no attributes found", async () => {
+ const response = await scrape({
+ url: "https://httpbin.org/html",
+ formats: [
+ {
+ type: "attributes",
+ selectors: [
+ { selector: ".nonexistent", attribute: "data-test" }
+ ]
+ }
+ ]
+ }, identity);
+
+ expect(response.attributes).toBeDefined();
+ expect(Array.isArray(response.attributes)).toBe(true);
+ expect(response.attributes!.length).toBe(1);
+ expect(response.attributes![0].values).toEqual([]);
+ }, scrapeTimeout);
+});
diff --git a/apps/api/src/controllers/v2/types.ts b/apps/api/src/controllers/v2/types.ts
index d7222bf01..6d1c002dc 100644
--- a/apps/api/src/controllers/v2/types.ts
+++ b/apps/api/src/controllers/v2/types.ts
@@ -212,6 +212,17 @@ export const screenshotFormatWithOptions = z.object({
export type ScreenshotFormatWithOptions = z.output;
+export const attributesFormatWithOptions = z.object({
+ type: z.literal("attributes"),
+ selectors: z.array(z.object({
+ selector: z.string().describe("CSS selector to find elements"),
+ attribute: z.string().describe("Attribute name to extract (e.g., 'data-vehicle-name' or 'id')")
+ })).describe("Extract specific attributes from elements"),
+}).strict();
+
+export type AttributesFormatWithOptions = z.output;
+
+
export type FormatObject =
| { type: "markdown" }
| { type: "html" }
@@ -221,7 +232,8 @@ export type FormatObject =
| { type: "summary" }
| JsonFormatWithOptions
| ChangeTrackingFormatWithOptions
- | ScreenshotFormatWithOptions;
+ | ScreenshotFormatWithOptions
+ | AttributesFormatWithOptions
export const parsersSchema = z.array(z.enum(["pdf"])).default(["pdf"]);
@@ -257,6 +269,7 @@ const baseScrapeOptions = z
jsonFormatWithOptions,
changeTrackingFormatWithOptions,
screenshotFormatWithOptions,
+ attributesFormatWithOptions,
])
.array()
.optional()
@@ -638,6 +651,11 @@ export type Document = {
json?: any;
summary?: string;
warning?: string;
+ attributes?: {
+ selector: string;
+ attribute: string;
+ values: string[];
+ }[];
actions?: {
screenshots?: string[];
scrapes?: ScrapeActionContent[];
diff --git a/apps/api/src/lib/html-transformer.ts b/apps/api/src/lib/html-transformer.ts
index 68bf6eb4e..4f05fe78c 100644
--- a/apps/api/src/lib/html-transformer.ts
+++ b/apps/api/src/lib/html-transformer.ts
@@ -28,6 +28,7 @@ class RustHTMLTransformer {
private _transformHtml: KoffiFunction;
private _freeString: KoffiFunction;
private _getInnerJSON: KoffiFunction;
+ private _extractAttributes: KoffiFunction;
private constructor() {
const lib = koffi.load(rustExecutablePath);
@@ -40,6 +41,7 @@ class RustHTMLTransformer {
this._extractMetadata = lib.func("extract_metadata", freedResultString, ["string"]);
this._transformHtml = lib.func("transform_html", freedResultString, ["string"]);
this._getInnerJSON = lib.func("get_inner_json", freedResultString, ["string"]);
+ this._extractAttributes = lib.func("extract_attributes", freedResultString, ["string", "string"]);
}
public static async getInstance(): Promise {
@@ -137,6 +139,22 @@ class RustHTMLTransformer {
});
});
}
+
+ public async extractAttributes(html: string, options: string): Promise {
+ return new Promise((resolve, reject) => {
+ this._extractAttributes.async(html, options, (err: Error, res: string) => {
+ if (err) {
+ reject(err);
+ } else {
+ if (res.startsWith("RUSTFC:ERROR:")) {
+ reject(new Error("Rust attribute extraction failed: " + res.split("RUSTFC:ERROR:")[1]));
+ } else {
+ resolve(res);
+ }
+ }
+ });
+ });
+ }
}
export async function extractLinks(
@@ -198,3 +216,33 @@ export async function getInnerJSON(
const converter = await RustHTMLTransformer.getInstance();
return await converter.getInnerJSON(html);
}
+
+export type AttributeSelector = {
+ selector: string;
+ attribute: string;
+};
+
+export type AttributeResult = {
+ selector: string;
+ attribute: string;
+ values: string[];
+};
+
+export async function extractAttributesRust(
+ html: string,
+ selectors: AttributeSelector[]
+): Promise {
+ if (!html || selectors.length === 0) {
+ return [];
+ }
+
+ const converter = await RustHTMLTransformer.getInstance();
+ const options = JSON.stringify({ selectors });
+ const resultJson = await converter.extractAttributes(html, options);
+
+ try {
+ return JSON.parse(resultJson);
+ } catch (error) {
+ throw new Error(`Failed to parse Rust attribute extraction result: ${error}`);
+ }
+}
diff --git a/apps/api/src/scraper/scrapeURL/lib/extractAttributes.ts b/apps/api/src/scraper/scrapeURL/lib/extractAttributes.ts
new file mode 100644
index 000000000..89480a1d5
--- /dev/null
+++ b/apps/api/src/scraper/scrapeURL/lib/extractAttributes.ts
@@ -0,0 +1,101 @@
+import { load } from "cheerio";
+import { logger } from "../../../lib/logger";
+import { extractAttributesRust } from "../../../lib/html-transformer";
+
+export type AttributeResult = {
+ selector: string;
+ attribute: string;
+ values: string[];
+};
+
+export type AttributeSelector = {
+ selector: string;
+ attribute: string;
+};
+
+/**
+ * Extracts attributes from HTML using Rust html-transformer (with Cheerio fallback)
+ * @param html - The HTML content to extract from
+ * @param selectors - Array of selector/attribute pairs to extract
+ * @returns Array of extracted attribute results
+ */
+export async function extractAttributes(
+ html: string,
+ selectors: AttributeSelector[]
+): Promise {
+ if (!selectors || selectors.length === 0) {
+ return [];
+ }
+
+ // Try Rust implementation first (faster, non-blocking)
+ try {
+ const results = await extractAttributesRust(html, selectors);
+
+ logger.debug("Attribute extraction via Rust", {
+ selectorsCount: selectors.length,
+ resultsCount: results.length,
+ sampleResults: results.slice(0, 2).map(r => ({
+ selector: r.selector,
+ attribute: r.attribute,
+ valuesCount: r.values.length
+ }))
+ });
+
+ return results;
+ } catch (error) {
+ logger.warn("Failed to extract attributes with Rust, falling back to Cheerio", {
+ error,
+ module: "scrapeURL",
+ method: "extractAttributes"
+ });
+ }
+
+ // Fallback to Cheerio implementation
+ const results: AttributeResult[] = [];
+
+ try {
+ const $ = load(html);
+
+ for (const extraction of selectors) {
+ const { selector, attribute } = extraction;
+ const values: string[] = [];
+
+ // Find all elements matching the selector
+ $(selector).each((_, element) => {
+ // Get the attribute value
+ // Support both data-* format and without data- prefix
+ let attrValue = $(element).attr(attribute);
+
+ // If not found and attribute doesn't start with 'data-', try with 'data-' prefix
+ if (!attrValue && !attribute.startsWith('data-')) {
+ attrValue = $(element).attr(`data-${attribute}`);
+ }
+
+ if (attrValue) {
+ values.push(attrValue);
+ }
+ });
+
+ results.push({
+ selector,
+ attribute,
+ values
+ });
+
+ logger.debug("Attribute extraction via Cheerio fallback", {
+ selector,
+ attribute,
+ valuesCount: values.length,
+ sample: values.slice(0, 3)
+ });
+ }
+ } catch (error) {
+ logger.error("Failed to extract attributes with Cheerio fallback", {
+ error,
+ module: "scrapeURL",
+ method: "extractAttributes"
+ });
+ }
+
+ return results;
+}
diff --git a/apps/api/src/scraper/scrapeURL/transformers/index.ts b/apps/api/src/scraper/scrapeURL/transformers/index.ts
index e3d293d08..4b7bfd4f1 100644
--- a/apps/api/src/scraper/scrapeURL/transformers/index.ts
+++ b/apps/api/src/scraper/scrapeURL/transformers/index.ts
@@ -9,6 +9,7 @@ import { performLLMExtract, performSummary } from "./llmExtract";
import { uploadScreenshot } from "./uploadScreenshot";
import { removeBase64Images } from "./removeBase64Images";
import { performAgent } from "./agent";
+import { performAttributes } from "./performAttributes";
import { deriveDiff } from "./diff";
import { useIndex } from "../../../services/index";
@@ -305,6 +306,7 @@ export const transformerStack: Transformer[] = [
...(useIndex ? [sendDocumentToIndex] : []),
performLLMExtract,
performSummary,
+ performAttributes,
performAgent,
deriveDiff,
coerceFieldsToFormats,
diff --git a/apps/api/src/scraper/scrapeURL/transformers/performAttributes.ts b/apps/api/src/scraper/scrapeURL/transformers/performAttributes.ts
new file mode 100644
index 000000000..8e5881ec6
--- /dev/null
+++ b/apps/api/src/scraper/scrapeURL/transformers/performAttributes.ts
@@ -0,0 +1,50 @@
+import { Document } from "../../../controllers/v2/types";
+import { Meta } from "..";
+import { extractAttributes } from "../lib/extractAttributes";
+import { hasFormatOfType } from "../../../lib/format-utils";
+
+/**
+ * Transformer to extract attributes from HTML using the attributes format
+ */
+export async function performAttributes(
+ meta: Meta,
+ document: Document,
+): Promise {
+ const attributesFormat = hasFormatOfType(meta.options.formats, "attributes");
+
+ if (!attributesFormat) {
+ return document;
+ }
+
+ if (document.html === undefined) {
+ throw new Error(
+ "html is undefined -- this transformer is being called out of order",
+ );
+ }
+
+ if (attributesFormat.selectors && attributesFormat.selectors.length > 0) {
+ try {
+ const attributes = await extractAttributes(document.html, attributesFormat.selectors);
+
+ if (attributes.length > 0) {
+ document.attributes = attributes;
+
+ meta.logger.debug("Extracted attributes", {
+ count: attributes.length,
+ attributes: attributes.map(d => ({
+ selector: d.selector,
+ attribute: d.attribute,
+ valuesCount: d.values.length
+ }))
+ });
+ }
+ } catch (error) {
+ meta.logger.error("Failed to extract attributes", {
+ error,
+ url: meta.url
+ });
+ }
+ }
+
+ return document;
+}
diff --git a/apps/js-sdk/firecrawl/src/v2/types.ts b/apps/js-sdk/firecrawl/src/v2/types.ts
index 0e0d76e66..7ee12ac5b 100644
--- a/apps/js-sdk/firecrawl/src/v2/types.ts
+++ b/apps/js-sdk/firecrawl/src/v2/types.ts
@@ -10,7 +10,8 @@ export type FormatString =
| "screenshot"
| "summary"
| "changeTracking"
- | "json";
+ | "json"
+ | "attributes";
export interface Viewport {
width: number;
@@ -41,13 +42,21 @@ export interface ChangeTrackingFormat extends Format {
prompt?: string;
tag?: string;
}
+export interface AttributesFormat extends Format {
+ type: "attributes";
+ selectors: Array<{
+ selector: string;
+ attribute: string;
+ }>;
+}
export type FormatOption =
| FormatString
| Format
| JsonFormat
| ChangeTrackingFormat
- | ScreenshotFormat;
+ | ScreenshotFormat
+ | AttributesFormat;
export interface LocationConfig {
country?: string;
@@ -170,6 +179,11 @@ export interface Document {
links?: string[];
images?: string[];
screenshot?: string;
+ attributes?: Array<{
+ selector: string;
+ attribute: string;
+ values: string[];
+ }>;
actions?: Record;
warning?: string;
changeTracking?: Record;
diff --git a/apps/python-sdk/firecrawl/v2/types.py b/apps/python-sdk/firecrawl/v2/types.py
index 69e71f830..148362534 100644
--- a/apps/python-sdk/firecrawl/v2/types.py
+++ b/apps/python-sdk/firecrawl/v2/types.py
@@ -114,6 +114,12 @@ class DocumentMetadata(BaseModel):
def coerce_status_code_to_int(cls, v):
return cls._coerce_string_to_int(v)
+class AttributeResult(BaseModel):
+ """Result of attribute extraction."""
+ selector: str
+ attribute: str
+ values: List[str]
+
class Document(BaseModel):
"""A scraped document."""
markdown: Optional[str] = None
@@ -183,7 +189,7 @@ CategoryOption = Union[str, Category]
FormatString = Literal[
# camelCase versions (API format)
- "markdown", "html", "rawHtml", "links", "images", "screenshot", "summary", "changeTracking", "json",
+ "markdown", "html", "rawHtml", "links", "images", "screenshot", "summary", "changeTracking", "json", "attributes",
# snake_case versions (user-friendly)
"raw_html", "change_tracking"
]
@@ -215,9 +221,18 @@ class ScreenshotFormat(BaseModel):
full_page: Optional[bool] = None
quality: Optional[int] = None
viewport: Optional[Union[Dict[str, int], Viewport]] = None
+
+class AttributeSelector(BaseModel):
+ """Selector and attribute pair for attribute extraction."""
+ selector: str
+ attribute: str
-FormatOption = Union[Dict[str, Any], FormatString, JsonFormat, ChangeTrackingFormat, ScreenshotFormat, Format]
+class AttributesFormat(Format):
+ """Configuration for attribute extraction."""
+ type: Literal["attributes"] = "attributes"
+ selectors: List[AttributeSelector]
+FormatOption = Union[Dict[str, Any], FormatString, JsonFormat, ChangeTrackingFormat, ScreenshotFormat, AttributesFormat, Format]
# Scrape types
class ScrapeFormats(BaseModel):
"""Output formats for scraping."""
diff --git a/examples/attributes-extraction-js-sdk.js b/examples/attributes-extraction-js-sdk.js
new file mode 100644
index 000000000..eb2651f43
--- /dev/null
+++ b/examples/attributes-extraction-js-sdk.js
@@ -0,0 +1,54 @@
+/**
+ * Example: Using Firecrawl JS SDK v2 to extract attributes from HTML elements
+ */
+
+import FirecrawlApp from '@mendable/firecrawl-js';
+
+const app = new FirecrawlApp({ apiKey: process.env.FIRECRAWL_API_KEY });
+
+async function main() {
+ console.log('šÆ Extracting attributes from Hacker News...');
+
+ try {
+ // Extract story IDs from Hacker News
+ const result = await app.scrapeUrl('https://news.ycombinator.com', {
+ formats: [
+ { type: 'markdown' },
+ {
+ type: 'attributes',
+ selectors: [
+ { selector: '.athing', attribute: 'id' }
+ ]
+ }
+ ]
+ });
+
+ console.log('ā
Success! Extracted data:');
+ console.log('Story IDs:', result.data.attributes[0].values.slice(0, 5));
+ console.log('Total stories found:', result.data.attributes[0].values.length);
+
+ // Example with GitHub - multiple attributes
+ console.log('\nšÆ Extracting multiple attributes from GitHub...');
+
+ const githubResult = await app.scrapeUrl('https://github.com/microsoft/vscode', {
+ formats: [
+ {
+ type: 'attributes',
+ selectors: [
+ { selector: '[data-testid]', attribute: 'data-testid' },
+ { selector: '[data-view-component]', attribute: 'data-view-component' }
+ ]
+ }
+ ]
+ });
+
+ console.log('ā
GitHub extraction success!');
+ console.log('Test IDs found:', githubResult.data.attributes[0].values.length);
+ console.log('Components found:', githubResult.data.attributes[1].values.length);
+
+ } catch (error) {
+ console.error('ā Error:', error.message);
+ }
+}
+
+main();
\ No newline at end of file
diff --git a/examples/attributes-extraction-python-sdk.py b/examples/attributes-extraction-python-sdk.py
new file mode 100644
index 000000000..1f5512a89
--- /dev/null
+++ b/examples/attributes-extraction-python-sdk.py
@@ -0,0 +1,60 @@
+"""
+Example: Using Firecrawl Python SDK v2 to extract attributes from HTML elements
+"""
+
+import os
+from firecrawl import FirecrawlApp
+
+def main():
+ app = FirecrawlApp(api_key=os.getenv('FIRECRAWL_API_KEY'))
+
+ print('šÆ Extracting attributes from Hacker News...')
+
+ try:
+ # Extract story IDs from Hacker News
+ result = app.scrape_url('https://news.ycombinator.com', {
+ 'formats': [
+ {'type': 'markdown'},
+ {
+ 'type': 'attributes',
+ 'selectors': [
+ {'selector': '.athing', 'attribute': 'id'}
+ ]
+ }
+ ]
+ })
+
+ if result.get('attributes'):
+ story_ids = result['attributes'][0]['values']
+ print(f'ā
Success! Found {len(story_ids)} stories')
+ print(f'Sample story IDs: {story_ids[:5]}')
+
+ # Example with GitHub - multiple attributes
+ print('\nšÆ Extracting multiple attributes from GitHub...')
+
+ github_result = app.scrape_url('https://github.com/microsoft/vscode', {
+ 'formats': [
+ {
+ 'type': 'attributes',
+ 'selectors': [
+ {'selector': '[data-testid]', 'attribute': 'data-testid'},
+ {'selector': '[data-view-component]', 'attribute': 'data-view-component'}
+ ]
+ }
+ ]
+ })
+
+ if github_result.get('attributes'):
+ test_ids = github_result['attributes'][0]['values']
+ components = github_result['attributes'][1]['values']
+
+ print(f'ā
GitHub extraction success!')
+ print(f'Test IDs found: {len(test_ids)}')
+ print(f'Components found: {len(components)}')
+ print(f'Sample test IDs: {test_ids[:3]}')
+
+ except Exception as error:
+ print(f'ā Error: {error}')
+
+if __name__ == '__main__':
+ main()
\ No newline at end of file