From d2b325f815b7829f4ea1bf08948ac81781026ff7 Mon Sep 17 00:00:00 2001 From: rafaelmmiller <150964962+rafaelsideguide@users.noreply.github.com> Date: Thu, 7 Aug 2025 15:52:07 -0300 Subject: [PATCH] (python-sdk): get_crawl_errors and active_crawls, got rid of useless tests --- apps/api/requests/v2/crawl.requests.http | 7 + apps/python-sdk/example.py | 40 +++- .../firecrawl/__tests__/e2e/v2/test_crawl.py | 84 +++++++ .../v2/methods/crawl/test_crawl_params.py | 224 ++++-------------- .../unit/v2/methods/crawl/test_webhook.py | 123 ++++++++++ .../unit/v2/utils/test_validation.py | 8 +- apps/python-sdk/firecrawl/client.py | 3 + apps/python-sdk/firecrawl/types.py | 8 +- apps/python-sdk/firecrawl/v2/__init__.py | 28 +-- apps/python-sdk/firecrawl/v2/client.py | 211 ++++++++++++++++- apps/python-sdk/firecrawl/v2/methods/crawl.py | 76 +++++- apps/python-sdk/firecrawl/v2/types.py | 55 +++-- 12 files changed, 634 insertions(+), 233 deletions(-) create mode 100644 apps/python-sdk/firecrawl/__tests__/unit/v2/methods/crawl/test_webhook.py diff --git a/apps/api/requests/v2/crawl.requests.http b/apps/api/requests/v2/crawl.requests.http index 47d34fdd2..72be878ee 100644 --- a/apps/api/requests/v2/crawl.requests.http +++ b/apps/api/requests/v2/crawl.requests.http @@ -48,3 +48,10 @@ content-type: application/json "url": "https://firecrawl.dev", "includePaths": ["blog", "news"] } + + +### Active Crawls +# @name activeCrawls +GET {{baseUrl}}/v2/crawl/active HTTP/1.1 +Authorization: Bearer {{$dotenv TEST_API_KEY}} +content-type: application/json \ No newline at end of file diff --git a/apps/python-sdk/example.py b/apps/python-sdk/example.py index 8cf3147dc..848a55ef5 100644 --- a/apps/python-sdk/example.py +++ b/apps/python-sdk/example.py @@ -7,7 +7,7 @@ import os import time from dotenv import load_dotenv from firecrawl import Firecrawl -from firecrawl.v2.types import ScrapeOptions, ScrapeFormats +from firecrawl.v2.types import ScrapeOptions, ScrapeFormats, WebhookConfig load_dotenv() @@ -22,18 +22,23 @@ def main(): firecrawl = Firecrawl(api_key=api_key, api_url=api_url) - # crawl + # crawl crawl_response = firecrawl.crawl("docs.firecrawl.dev", limit=5) print(crawl_response) + # start crawl crawl_job = firecrawl.start_crawl('docs.firecrawl.dev', limit=5) print(crawl_job) - while (crawl_job.status != 'completed'): - crawl_job = firecrawl.get_crawl_status(crawl_job.id) + crawl_response = firecrawl.get_crawl_status(crawl_job.id) + print(crawl_response) + + while (crawl_response.status != 'completed'): + print(f"Crawl status: {crawl_response.status}") + crawl_response = firecrawl.get_crawl_status(crawl_job.id) time.sleep(2) - print(crawl_job) + print(crawl_response) # crawl params preview params_data = firecrawl.crawl_params_preview( @@ -42,6 +47,31 @@ def main(): ) print(params_data) + # crawl with webhook example + webhook_job = firecrawl.start_crawl( + "docs.firecrawl.dev", + limit=3, + webhook="https://your-webhook-endpoint.com/firecrawl" + ) + + # advanced webhook with configuration + webhook_config = WebhookConfig( + url="https://your-webhook-endpoint.com/firecrawl", + headers={"Authorization": "Bearer your-token"}, + events=["completed", "failed"] + ) + + webhook_job_advanced = firecrawl.start_crawl( + "docs.firecrawl.dev", + limit=2, + webhook=webhook_config + ) + + # Check crawl errors + errors = firecrawl.get_crawl_errors(crawl_job.id) + print(f"Crawl errors: {errors.errors}") + print(f"Robots blocked: {errors.robots_blocked}") + # search examples search_response = firecrawl.search( query="What is the capital of France?", diff --git a/apps/python-sdk/firecrawl/__tests__/e2e/v2/test_crawl.py b/apps/python-sdk/firecrawl/__tests__/e2e/v2/test_crawl.py index 47db5bded..3a70e5e55 100644 --- a/apps/python-sdk/firecrawl/__tests__/e2e/v2/test_crawl.py +++ b/apps/python-sdk/firecrawl/__tests__/e2e/v2/test_crawl.py @@ -77,6 +77,90 @@ class TestCrawlE2E: time.sleep(5) assert cancel_job == True + def test_get_crawl_errors(self): + """Test getting crawl errors.""" + # First start a crawl + start_job = self.client.start_crawl("https://docs.firecrawl.dev", limit=3) + assert start_job.id is not None + + job_id = start_job.id + + # Get errors (should work even if no errors exist) + errors_response = self.client.get_crawl_errors(job_id) + + # Verify the response structure + assert hasattr(errors_response, 'errors') + assert hasattr(errors_response, 'robots_blocked') + assert isinstance(errors_response.errors, list) + assert isinstance(errors_response.robots_blocked, list) + + # Errors list should contain dictionaries with expected fields + for error in errors_response.errors: + assert isinstance(error, dict) + assert 'id' in error + assert 'timestamp' in error + assert 'url' in error + assert 'error' in error + assert isinstance(error['id'], str) + assert isinstance(error['timestamp'], str) + assert isinstance(error['url'], str) + assert isinstance(error['error'], str) + + # Robots blocked should be a list of strings + for blocked_url in errors_response.robots_blocked: + assert isinstance(blocked_url, str) + + def test_get_crawl_errors_with_invalid_job_id(self): + """Test getting crawl errors with an invalid job ID.""" + with pytest.raises(Exception): + self.client.get_crawl_errors("invalid-job-id-12345") + + def test_get_active_crawls(self): + """Test getting active crawls.""" + # Get active crawls + active_crawls_response = self.client.active_crawls() + + # Verify the response structure + assert hasattr(active_crawls_response, 'success') + assert hasattr(active_crawls_response, 'crawls') + assert isinstance(active_crawls_response.success, bool) + assert isinstance(active_crawls_response.crawls, list) + + # Each crawl should have the required fields + for crawl in active_crawls_response.crawls: + assert hasattr(crawl, 'id') + assert hasattr(crawl, 'team_id') + assert hasattr(crawl, 'url') + assert isinstance(crawl.id, str) + assert isinstance(crawl.team_id, str) + assert isinstance(crawl.url, str) + + # Options field is optional but if present should be a dict + if hasattr(crawl, 'options') and crawl.options is not None: + assert isinstance(crawl.options, dict) + + def test_get_active_crawls_with_running_crawl(self): + """Test getting active crawls when there's a running crawl.""" + # Start a crawl + start_job = self.client.start_crawl("https://docs.firecrawl.dev", limit=5) + assert start_job.id is not None + + # Get active crawls + active_crawls_response = self.client.active_crawls() + + # Verify the response structure + assert hasattr(active_crawls_response, 'success') + assert hasattr(active_crawls_response, 'crawls') + assert isinstance(active_crawls_response.success, bool) + assert isinstance(active_crawls_response.crawls, list) + + # The started crawl should be in the active crawls list + active_crawl_ids = [crawl.id for crawl in active_crawls_response.crawls] + assert start_job.id in active_crawl_ids + + # Cancel the crawl to clean up + self.client.cancel_crawl(start_job.id) + def test_crawl_with_wait(self): """Test crawl with wait for completion.""" crawl_job = self.client.crawl( diff --git a/apps/python-sdk/firecrawl/__tests__/unit/v2/methods/crawl/test_crawl_params.py b/apps/python-sdk/firecrawl/__tests__/unit/v2/methods/crawl/test_crawl_params.py index 388b259a0..039870fb8 100644 --- a/apps/python-sdk/firecrawl/__tests__/unit/v2/methods/crawl/test_crawl_params.py +++ b/apps/python-sdk/firecrawl/__tests__/unit/v2/methods/crawl/test_crawl_params.py @@ -1,196 +1,70 @@ +""" +Unit tests for crawl params functionality in Firecrawl v2 SDK. +""" + import pytest -from unittest.mock import Mock, patch -from firecrawl.v2.types import CrawlParamsRequest, CrawlParamsResponse, CrawlParamsData -from firecrawl.v2.methods.crawl import crawl_params_preview +from firecrawl.v2.types import CrawlParamsRequest, CrawlParamsData -class TestCrawlParams: - """Unit tests for crawl_params function.""" +class TestCrawlParamsRequest: + """Unit tests for CrawlParamsRequest.""" - def test_crawl_params_success(self): - """Test successful crawl_params call.""" - # Mock client and response - mock_client = Mock() - mock_response = Mock() - mock_response.ok = True - mock_response.json.return_value = { - "success": True, - "data": { - "limit": 10, - "maxDiscoveryDepth": 3, - "ignoreSitemap": False - }, - "warning": None - } - mock_client.post.return_value = mock_response - - # Create request + def test_crawl_params_request_creation(self): + """Test creating CrawlParamsRequest with valid data.""" request = CrawlParamsRequest( url="https://example.com", prompt="Extract all blog posts" ) - # Call function - result = crawl_params_preview(mock_client, request) - - # Verify client call - mock_client.post.assert_called_once_with("/v2/crawl-params", { - "url": "https://example.com", - "prompt": "Extract all blog posts" - }) - - # Verify result - assert isinstance(result, CrawlParamsData) - assert result.limit == 10 - assert result.max_discovery_depth == 3 - assert result.ignore_sitemap is False - assert result.warning is None + assert request.url == "https://example.com" + assert request.prompt == "Extract all blog posts" - def test_crawl_params_api_error(self): - """Test crawl_params with API error.""" - # Mock client and response - mock_client = Mock() - mock_response = Mock() - mock_response.ok = False - mock_response.status_code = 400 - mock_response.text = "Bad Request" - mock_client.post.return_value = mock_response - - # Create request + def test_crawl_params_request_serialization(self): + """Test that CrawlParamsRequest serializes correctly.""" request = CrawlParamsRequest( url="https://example.com", - prompt="Extract all blog posts" + prompt="Extract all blog posts and documentation" ) - # Call function and expect exception - with pytest.raises(Exception, match="crawl params"): - crawl_params_preview(mock_client, request) + data = request.model_dump() + + assert data["url"] == "https://example.com" + assert data["prompt"] == "Extract all blog posts and documentation" - def test_crawl_params_success_false(self): - """Test crawl_params with success=False in response.""" - # Mock client and response - mock_client = Mock() - mock_response = Mock() - mock_response.ok = True - mock_response.json.return_value = { - "success": False, - "error": "Invalid URL provided" - } - mock_client.post.return_value = mock_response - - # Create request - request = CrawlParamsRequest( - url="https://example.com", - prompt="Extract all blog posts" - ) - - # Call function and expect exception - with pytest.raises(Exception, match="Invalid URL provided"): - crawl_params_preview(mock_client, request) - def test_crawl_params_empty_url(self): - """Test crawl_params with empty URL.""" - # Create request with empty URL - request = CrawlParamsRequest( - url="", - prompt="Extract all blog posts" - ) - - # Call function and expect exception - with pytest.raises(ValueError, match="URL cannot be empty"): - crawl_params_preview(Mock(), request) +class TestCrawlParamsData: + """Unit tests for CrawlParamsData.""" - def test_crawl_params_whitespace_url(self): - """Test crawl_params with whitespace-only URL.""" - # Create request with whitespace URL - request = CrawlParamsRequest( - url=" ", - prompt="Extract all blog posts" - ) + def test_crawl_params_data_creation(self): + """Test creating CrawlParamsData with minimal data.""" + data = CrawlParamsData() - # Call function and expect exception - with pytest.raises(ValueError, match="URL cannot be empty"): - crawl_params_preview(Mock(), request) + assert data.include_paths is None + assert data.exclude_paths is None + assert data.max_discovery_depth is None + assert data.ignore_sitemap is False + assert data.limit is None + assert data.crawl_entire_domain is False + assert data.allow_external_links is False + assert data.scrape_options is None + assert data.warning is None - def test_crawl_params_empty_prompt(self): - """Test crawl_params with empty prompt.""" - # Create request with empty prompt - request = CrawlParamsRequest( - url="https://example.com", - prompt="" + def test_crawl_params_data_with_values(self): + """Test creating CrawlParamsData with values.""" + data = CrawlParamsData( + include_paths=["/blog/*"], + exclude_paths=["/admin/*"], + max_discovery_depth=3, + limit=50, + crawl_entire_domain=True, + allow_external_links=False, + warning="Test warning" ) - # Call function and expect exception - with pytest.raises(ValueError, match="Prompt cannot be empty"): - crawl_params_preview(Mock(), request) - - def test_crawl_params_whitespace_prompt(self): - """Test crawl_params with whitespace-only prompt.""" - # Create request with whitespace prompt - request = CrawlParamsRequest( - url="https://example.com", - prompt=" " - ) - - # Call function and expect exception - with pytest.raises(ValueError, match="Prompt cannot be empty"): - crawl_params_preview(Mock(), request) - - def test_crawl_params_complex_options(self): - """Test crawl_params with complex options in response.""" - # Mock client and response - mock_client = Mock() - mock_response = Mock() - mock_response.ok = True - mock_response.json.return_value = { - "success": True, - "data": { - "includePaths": ["/blog/*", "/docs/*"], - "excludePaths": ["/admin/*"], - "maxDiscoveryDepth": 3, - "ignoreSitemap": False, - "limit": 50, - "crawlEntireDomain": True, - "allowExternalLinks": False, - "scrapeOptions": { - "formats": ["markdown"], - "onlyMainContent": False, - "mobile": True, - "timeout": None, - "waitFor": None, - "skipTlsVerification": False, - "removeBase64Images": True - } - } - } - mock_client.post.return_value = mock_response - - # Create request - request = CrawlParamsRequest( - url="https://example.com", - prompt="Extract all blog posts and documentation with mobile view" - ) - - # Call function - result = crawl_params_preview(mock_client, request) - - # Verify result - assert result is not None - - # Check all fields - assert result.include_paths == ["/blog/*", "/docs/*"] - assert result.exclude_paths == ["/admin/*"] - assert result.max_discovery_depth == 3 - assert result.ignore_sitemap is False - assert result.limit == 50 - assert result.crawl_entire_domain is True - assert result.allow_external_links is False - - # Check nested scrape options - assert result.scrape_options is not None - assert result.scrape_options.formats is not None - # formats is a ScrapeFormats object, so we need to access its formats field - assert len(result.scrape_options.formats.formats) == 1 - assert result.scrape_options.formats.formats[0].type == "markdown" - assert result.scrape_options.only_main_content is False - assert result.scrape_options.mobile is True \ No newline at end of file + assert data.include_paths == ["/blog/*"] + assert data.exclude_paths == ["/admin/*"] + assert data.max_discovery_depth == 3 + assert data.limit == 50 + assert data.crawl_entire_domain is True + assert data.allow_external_links is False + assert data.warning == "Test warning" \ No newline at end of file diff --git a/apps/python-sdk/firecrawl/__tests__/unit/v2/methods/crawl/test_webhook.py b/apps/python-sdk/firecrawl/__tests__/unit/v2/methods/crawl/test_webhook.py new file mode 100644 index 000000000..eca5d06fa --- /dev/null +++ b/apps/python-sdk/firecrawl/__tests__/unit/v2/methods/crawl/test_webhook.py @@ -0,0 +1,123 @@ +""" +Unit tests for webhook functionality in Firecrawl v2 SDK. +""" + +import pytest +from firecrawl.v2.types import WebhookConfig, CrawlRequest +from firecrawl.v2.methods.crawl import _prepare_crawl_request + +class TestWebhookConfig: + """Test WebhookConfig class functionality.""" + + def test_webhook_config_creation_minimal(self): + """Test creating WebhookConfig with minimal parameters.""" + webhook = WebhookConfig(url="https://example.com/webhook") + assert webhook.url == "https://example.com/webhook" + assert webhook.headers is None + assert webhook.metadata is None + assert webhook.events is None + + def test_webhook_config_creation_full(self): + """Test creating WebhookConfig with all parameters.""" + webhook = WebhookConfig( + url="https://example.com/webhook", + headers={"Authorization": "Bearer token"}, + metadata={"project": "test"}, + events=["completed", "failed"] + ) + assert webhook.url == "https://example.com/webhook" + assert webhook.headers == {"Authorization": "Bearer token"} + assert webhook.metadata == {"project": "test"} + assert webhook.events == ["completed", "failed"] + + def test_webhook_config_validation(self): + """Test WebhookConfig validation.""" + # URL is required + with pytest.raises(Exception): # Pydantic validation error + WebhookConfig() + + +class TestCrawlRequestWebhook: + """Test CrawlRequest webhook functionality.""" + + def test_crawl_request_with_string_webhook(self): + """Test CrawlRequest with string webhook.""" + request = CrawlRequest( + url="https://example.com", + webhook="https://example.com/webhook" + ) + + data = _prepare_crawl_request(request) + assert data["webhook"] == "https://example.com/webhook" + + def test_crawl_request_with_webhook_config(self): + """Test CrawlRequest with WebhookConfig object.""" + webhook_config = WebhookConfig( + url="https://example.com/webhook", + headers={"Authorization": "Bearer token"}, + events=["completed"] + ) + + request = CrawlRequest( + url="https://example.com", + webhook=webhook_config + ) + + data = _prepare_crawl_request(request) + assert data["webhook"]["url"] == "https://example.com/webhook" + assert data["webhook"]["headers"] == {"Authorization": "Bearer token"} + assert data["webhook"]["events"] == ["completed"] + + def test_crawl_request_without_webhook(self): + """Test CrawlRequest without webhook.""" + request = CrawlRequest(url="https://example.com") + + data = _prepare_crawl_request(request) + assert "webhook" not in data + + def test_crawl_request_webhook_serialization(self): + """Test that webhook config is properly serialized.""" + webhook_config = WebhookConfig( + url="https://example.com/webhook", + headers={"Content-Type": "application/json"}, + metadata={"test": "value"}, + events=["page", "completed"] + ) + + request = CrawlRequest( + url="https://example.com", + webhook=webhook_config + ) + + data = _prepare_crawl_request(request) + webhook_data = data["webhook"] + + # Check that all fields are properly serialized + assert webhook_data["url"] == "https://example.com/webhook" + assert webhook_data["headers"] == {"Content-Type": "application/json"} + assert webhook_data["metadata"] == {"test": "value"} + assert webhook_data["events"] == ["page", "completed"] + + def test_crawl_request_webhook_with_none_values(self): + """Test webhook config with None values are excluded from serialization.""" + webhook_config = WebhookConfig( + url="https://example.com/webhook", + headers=None, + metadata=None, + events=None + ) + + request = CrawlRequest( + url="https://example.com", + webhook=webhook_config + ) + + data = _prepare_crawl_request(request) + webhook_data = data["webhook"] + + # Only url should be present, None values should be excluded + assert webhook_data["url"] == "https://example.com/webhook" + assert "headers" not in webhook_data + assert "metadata" not in webhook_data + assert "events" not in webhook_data + diff --git a/apps/python-sdk/firecrawl/__tests__/unit/v2/utils/test_validation.py b/apps/python-sdk/firecrawl/__tests__/unit/v2/utils/test_validation.py index 015c8f6b8..ed7253816 100644 --- a/apps/python-sdk/firecrawl/__tests__/unit/v2/utils/test_validation.py +++ b/apps/python-sdk/firecrawl/__tests__/unit/v2/utils/test_validation.py @@ -1,5 +1,5 @@ import pytest -from firecrawl.v2.types import ScrapeOptions +from firecrawl.v2.types import JsonFormat, ScrapeOptions from firecrawl.v2.utils.validation import validate_scrape_options, prepare_scrape_options @@ -220,10 +220,8 @@ class TestPrepareScrapeOptions: def test_format_schema_conversion(self): """Test that Format schema is properly handled.""" - from firecrawl.v2.types import Format - - # Create a Format object with schema - format_obj = Format( + # Create a JsonFormat object with schema + format_obj = JsonFormat( type="json", prompt="Extract product info", schema={"type": "object", "properties": {"name": {"type": "string"}}} diff --git a/apps/python-sdk/firecrawl/client.py b/apps/python-sdk/firecrawl/client.py index a99e9326d..2bd5f1125 100644 --- a/apps/python-sdk/firecrawl/client.py +++ b/apps/python-sdk/firecrawl/client.py @@ -137,6 +137,9 @@ class Firecrawl: self.crawl_params_preview = self._v2_client.crawl_params_preview self.get_crawl_status = self._v2_client.get_crawl_status self.cancel_crawl = self._v2_client.cancel_crawl + self.get_crawl_errors = self._v2_client.get_crawl_errors + self.active_crawls = self._v2_client.active_crawls + # self.batch_scrape = self._v2_client.batch_scrape # self.map = self._v2_client.map self.search = self._v2_client.search diff --git a/apps/python-sdk/firecrawl/types.py b/apps/python-sdk/firecrawl/types.py index 755b8e40a..0d7323084 100644 --- a/apps/python-sdk/firecrawl/types.py +++ b/apps/python-sdk/firecrawl/types.py @@ -27,6 +27,8 @@ from .v2.types import ( CrawlParamsRequest, CrawlParamsData, CrawlParamsResponse, + CrawlErrorsResponse, + ActiveCrawlsResponse, # Batch scrape types BatchScrapeRequest, @@ -44,6 +46,7 @@ from .v2.types import ( Source, SourceOption, Format, + JsonFormat, FormatOption, SearchRequest, SearchResult, @@ -63,7 +66,6 @@ from .v2.types import ( # Location and format types Location, - JsonFormat, # Error types ErrorDetails, @@ -102,6 +104,8 @@ __all__ = [ 'CrawlParamsRequest', 'CrawlParamsData', 'CrawlParamsResponse', + 'CrawlErrorsResponse', + 'ActiveCrawlsResponse', # Batch scrape types 'BatchScrapeRequest', @@ -119,6 +123,7 @@ __all__ = [ 'Source', 'SourceOption', 'Format', + 'JsonFormat', 'FormatOption', 'SearchRequest', 'SearchResult', @@ -138,7 +143,6 @@ __all__ = [ # Location and format types 'Location', - 'JsonFormat', # Error types 'ErrorDetails', diff --git a/apps/python-sdk/firecrawl/v2/__init__.py b/apps/python-sdk/firecrawl/v2/__init__.py index fad147bce..01ce7a5a7 100644 --- a/apps/python-sdk/firecrawl/v2/__init__.py +++ b/apps/python-sdk/firecrawl/v2/__init__.py @@ -2,7 +2,7 @@ from typing import Optional, List, Union, Dict, Callable, Literal, Any from ..types import ( SearchData, SearchResult, Document, ScrapeOptions, SearchRequest, SourceOption, FormatOption, CrawlRequest, CrawlJob, CrawlResponse, - CrawlParamsRequest, CrawlParamsData + CrawlParamsRequest, CrawlParamsData, CrawlErrorsResponse, ActiveCrawlsResponse ) from .methods.search import search as search_method from .methods.crawl import ( @@ -10,7 +10,9 @@ from .methods.crawl import ( start_crawl as start_crawl_method, cancel_crawl as cancel_crawl_method, get_crawl_status as get_crawl_status_method, - crawl_params_preview as crawl_params_preview_method + crawl_params_preview as crawl_params_preview_method, + get_crawl_errors as get_crawl_errors_method, + get_active_crawls as get_active_crawls_method ) from .utils.http_client import HttpClient @@ -20,7 +22,7 @@ class FirecrawlClient: """ def __init__(self, api_key: str = None, api_url: str = "https://api.firecrawl.dev"): - self.api_key = api_key or "placeholder-key" + self.api_key = api_key self.api_url = api_url self._client = HttpClient(api_key=self.api_key, api_url=self.api_url) @@ -143,7 +145,6 @@ class FirecrawlClient: ) return start_crawl_method(self._client, request) - # get-crawl-status def get_crawl_status( self, job_id: str @@ -151,7 +152,6 @@ class FirecrawlClient: """Get the status of a crawl job.""" return get_crawl_status_method(self._client, job_id) - # cancel-crawl def cancel_crawl( self, job_id: str @@ -159,7 +159,6 @@ class FirecrawlClient: """Cancel a running crawl job.""" return cancel_crawl_method(self._client, job_id) - # crawl-params-preview def crawl_params_preview( self, url: str, @@ -169,17 +168,18 @@ class FirecrawlClient: request = CrawlParamsRequest(url=url, prompt=prompt) return crawl_params_preview_method(self._client, request) - # get-active-crawls - def get_active_crawls( + def active_crawls( self - ): - pass + ) -> ActiveCrawlsResponse: + """Get a list of active crawl jobs.""" + return get_active_crawls_method(self._client) - # get-crawl-errors def get_crawl_errors( - self - ): - pass + self, + crawl_id: str + ) -> CrawlErrorsResponse: + """Get errors from a crawl job.""" + return get_crawl_errors_method(self._client, crawl_id) # map def map( diff --git a/apps/python-sdk/firecrawl/v2/client.py b/apps/python-sdk/firecrawl/v2/client.py index 113ca961e..718281aef 100644 --- a/apps/python-sdk/firecrawl/v2/client.py +++ b/apps/python-sdk/firecrawl/v2/client.py @@ -5,21 +5,19 @@ This module provides the main client class that orchestrates all v2 functionalit """ import os -from typing import Optional, List, Dict, Any, Callable +from typing import Optional, List, Dict, Any, Callable, Union from .types import ( ClientConfig, ScrapeOptions, CrawlOptions, MapOptions, ExtractOptions, ScrapeResponse, CrawlResponse, CrawlStatusResponse, BatchScrapeResponse, BatchScrapeStatusResponse, MapResponse, ExtractResponse, Document, - SearchRequest, SearchResponse + SearchRequest, SearchResponse, CrawlRequest, WebhookConfig, CrawlErrorsResponse, ActiveCrawlsResponse ) from .utils.http_client import HttpClient from .utils.error_handler import FirecrawlError -# Import functions from modules -from . import scrape as scrape_module -from . import crawl as crawl_module -from . import batch as batch_module -from . import search as search_module - +from .methods import scrape as scrape_module +from .methods import crawl as crawl_module +from .methods import batch as batch_module +from .methods import search as search_module class FirecrawlClient: """ @@ -108,7 +106,6 @@ class FirecrawlClient: Returns: SearchResponse containing the search results """ - # Build options object from individual parameters options = SearchRequest( limit=limit, tbs=tbs, @@ -118,4 +115,200 @@ class FirecrawlClient: ) return search_module.search(self.http_client, query, options) + + def crawl( + self, + url: str, + *, + prompt: Optional[str] = None, + exclude_paths: Optional[List[str]] = None, + include_paths: Optional[List[str]] = None, + max_discovery_depth: Optional[int] = None, + ignore_sitemap: bool = False, + ignore_query_parameters: bool = False, + limit: Optional[int] = None, + crawl_entire_domain: bool = False, + allow_external_links: bool = False, + allow_subdomains: bool = False, + delay: Optional[int] = None, + max_concurrency: Optional[int] = None, + webhook: Optional[Union[str, WebhookConfig]] = None, + scrape_options: Optional[ScrapeOptions] = None, + zero_data_retention: bool = False, + poll_interval: int = 2, + timeout: Optional[int] = None + ) -> CrawlStatusResponse: + """ + Start a crawl job and wait for it to complete. + + Args: + url: Target URL to start crawling from + prompt: Optional prompt to guide the crawl + exclude_paths: Patterns of URLs to exclude + include_paths: Patterns of URLs to include + max_discovery_depth: Maximum depth for finding new URLs + ignore_sitemap: Skip sitemap.xml processing + ignore_query_parameters: Ignore URL parameters + limit: Maximum pages to crawl + crawl_entire_domain: Follow parent directory links + allow_external_links: Follow external domain links + allow_subdomains: Follow subdomains + delay: Delay in seconds between scrapes + max_concurrency: Maximum number of concurrent scrapes + webhook: Webhook configuration for notifications + scrape_options: Page scraping configuration + zero_data_retention: Whether to delete data after 24 hours + poll_interval: Seconds between status checks + timeout: Maximum seconds to wait (None for no timeout) + + Returns: + CrawlStatusResponse when job completes + + Raises: + ValueError: If request is invalid + Exception: If the crawl fails to start or complete + TimeoutError: If timeout is reached + """ + request = CrawlRequest( + url=url, + prompt=prompt, + exclude_paths=exclude_paths, + include_paths=include_paths, + max_discovery_depth=max_discovery_depth, + ignore_sitemap=ignore_sitemap, + ignore_query_parameters=ignore_query_parameters, + limit=limit, + crawl_entire_domain=crawl_entire_domain, + allow_external_links=allow_external_links, + allow_subdomains=allow_subdomains, + delay=delay, + max_concurrency=max_concurrency, + webhook=webhook, + scrape_options=scrape_options, + zero_data_retention=zero_data_retention + ) + + return crawl_module.crawl( + self.http_client, + request, + poll_interval=poll_interval, + timeout=timeout + ) + + def start_crawl( + self, + url: str, + *, + prompt: Optional[str] = None, + exclude_paths: Optional[List[str]] = None, + include_paths: Optional[List[str]] = None, + max_discovery_depth: Optional[int] = None, + ignore_sitemap: bool = False, + ignore_query_parameters: bool = False, + limit: Optional[int] = None, + crawl_entire_domain: bool = False, + allow_external_links: bool = False, + allow_subdomains: bool = False, + delay: Optional[int] = None, + max_concurrency: Optional[int] = None, + webhook: Optional[Union[str, WebhookConfig]] = None, + scrape_options: Optional[ScrapeOptions] = None, + zero_data_retention: bool = False + ) -> CrawlResponse: + """ + Start an asynchronous crawl job. + + Args: + url: Target URL to start crawling from + prompt: Optional prompt to guide the crawl + exclude_paths: Patterns of URLs to exclude + include_paths: Patterns of URLs to include + max_discovery_depth: Maximum depth for finding new URLs + ignore_sitemap: Skip sitemap.xml processing + ignore_query_parameters: Ignore URL parameters + limit: Maximum pages to crawl + crawl_entire_domain: Follow parent directory links + allow_external_links: Follow external domain links + allow_subdomains: Follow subdomains + delay: Delay in seconds between scrapes + max_concurrency: Maximum number of concurrent scrapes + webhook: Webhook configuration for notifications + scrape_options: Page scraping configuration + zero_data_retention: Whether to delete data after 24 hours + + Returns: + CrawlResponse with job information + + Raises: + ValueError: If request is invalid + Exception: If the crawl operation fails to start + """ + request = CrawlRequest( + url=url, + prompt=prompt, + exclude_paths=exclude_paths, + include_paths=include_paths, + max_discovery_depth=max_discovery_depth, + ignore_sitemap=ignore_sitemap, + ignore_query_parameters=ignore_query_parameters, + limit=limit, + crawl_entire_domain=crawl_entire_domain, + allow_external_links=allow_external_links, + allow_subdomains=allow_subdomains, + delay=delay, + max_concurrency=max_concurrency, + webhook=webhook, + scrape_options=scrape_options, + zero_data_retention=zero_data_retention + ) + + return crawl_module.start_crawl(self.http_client, request) + + def get_crawl_status(self, job_id: str) -> CrawlStatusResponse: + """ + Get the status of a crawl job. + + Args: + job_id: ID of the crawl job + + Returns: + CrawlStatusResponse with current status and data + + Raises: + Exception: If the status check fails + """ + return crawl_module.get_crawl_status(self.http_client, job_id) + + def check_crawl_errors(self, crawl_id: str) -> CrawlErrorsResponse: + """ + Get errors from a crawl job. + + Args: + crawl_id: The ID of the crawl job + + Returns: + CrawlErrorsResponse containing errors and robots blocked URLs + """ + return crawl_module.check_crawl_errors(self.http_client, crawl_id) + + def get_active_crawls(self) -> ActiveCrawlsResponse: + """ + Get a list of currently active crawl jobs. + + Returns: + ActiveCrawlsResponse containing a list of active crawl jobs. + """ + return crawl_module.get_active_crawls(self.http_client) + + def cancel_crawl(self, crawl_id: str) -> bool: + """ + Cancel a crawl job. + + Args: + crawl_id: The ID of the crawl job to cancel + + Returns: + bool: True if the crawl was cancelled, False otherwise + """ + return crawl_module.cancel_crawl(self.http_client, crawl_id) \ No newline at end of file diff --git a/apps/python-sdk/firecrawl/v2/methods/crawl.py b/apps/python-sdk/firecrawl/v2/methods/crawl.py index c86c85302..1894723f1 100644 --- a/apps/python-sdk/firecrawl/v2/methods/crawl.py +++ b/apps/python-sdk/firecrawl/v2/methods/crawl.py @@ -3,11 +3,12 @@ Crawling functionality for Firecrawl v2 API. """ import time -from typing import Optional +from typing import Optional, Dict, Any from ..types import ( CrawlRequest, CrawlJob, - CrawlResponse, Document, CrawlParamsRequest, CrawlParamsResponse, CrawlParamsData + CrawlResponse, Document, CrawlParamsRequest, CrawlParamsResponse, CrawlParamsData, + WebhookConfig, CrawlErrorsResponse, ActiveCrawlsResponse ) from ..utils import HttpClient, handle_response_error, validate_scrape_options, prepare_scrape_options @@ -67,6 +68,14 @@ def _prepare_crawl_request(request: CrawlRequest) -> dict: request_data.pop("prompt", None) request_data.pop("scrape_options", None) + # Handle webhook conversion first (before model_dump) + if request.webhook is not None: + if isinstance(request.webhook, str): + data["webhook"] = request.webhook + else: + # Convert WebhookConfig to dict + data["webhook"] = request.webhook.model_dump(exclude_none=True) + # Convert other snake_case fields to camelCase field_mappings = { "include_paths": "includePaths", @@ -79,7 +88,6 @@ def _prepare_crawl_request(request: CrawlRequest) -> dict: "allow_subdomains": "allowSubdomains", "delay": "delay", "max_concurrency": "maxConcurrency", - "webhook": "webhook", "zero_data_retention": "zeroDataRetention" } @@ -333,6 +341,14 @@ def crawl_params_preview(client: HttpClient, request: CrawlParamsRequest) -> Cra "zeroDataRetention": "zero_data_retention" } + # Handle webhook conversion + if "webhook" in params_data: + webhook_data = params_data["webhook"] + if isinstance(webhook_data, dict): + converted_params["webhook"] = WebhookConfig(**webhook_data) + else: + converted_params["webhook"] = webhook_data + for camel_case, snake_case in field_mappings.items(): if camel_case in params_data: if camel_case == "scrapeOptions" and params_data[camel_case] is not None: @@ -382,4 +398,56 @@ def crawl_params_preview(client: HttpClient, request: CrawlParamsRequest) -> Cra return CrawlParamsData(**converted_params) else: - raise Exception(response_data.get("error", "Unknown error occurred")) \ No newline at end of file + raise Exception(response_data.get("error", "Unknown error occurred")) + + +def get_crawl_errors(http_client: HttpClient, crawl_id: str) -> CrawlErrorsResponse: + """ + Get errors from a crawl job. + + Args: + http_client: HTTP client for making requests + crawl_id: The ID of the crawl job + + Returns: + CrawlErrorsResponse containing errors and robots blocked URLs + + Raises: + Exception: If the request fails + """ + response = http_client.get(f"/v2/crawl/{crawl_id}/errors") + + if response.status_code != 200: + handle_response_error(response, "check crawl errors") + + try: + data = response.json() + return CrawlErrorsResponse(**data) + except Exception as e: + raise Exception(f"Failed to parse crawl errors response: {e}") + + +def get_active_crawls(client: HttpClient) -> ActiveCrawlsResponse: + """ + Get a list of currently active crawl jobs. + + Args: + client: HTTP client instance + + Returns: + ActiveCrawlsResponse containing a list of active crawl jobs + + Raises: + Exception: If the request fails + """ + response = client.get("/v2/crawl/active") + + if not response.ok: + handle_response_error(response, "get active crawls") + + response_data = response.json() + + if response_data.get("success"): + return ActiveCrawlsResponse(**response_data.get("data", {})) + else: + raise Exception(response_data.get("error", "Unknown error occurred")) diff --git a/apps/python-sdk/firecrawl/v2/types.py b/apps/python-sdk/firecrawl/v2/types.py index c3e461c66..b4262314f 100644 --- a/apps/python-sdk/firecrawl/v2/types.py +++ b/apps/python-sdk/firecrawl/v2/types.py @@ -54,13 +54,29 @@ class Document(BaseModel): warning: Optional[str] = None change_tracking: Optional[Dict[str, Any]] = Field(None, alias="changeTracking") +# Webhook types +class WebhookConfig(BaseModel): + """Configuration for webhooks.""" + url: str + headers: Optional[Dict[str, str]] = None + metadata: Optional[Dict[str, str]] = None + events: Optional[List[Literal["completed", "failed", "page", "started"]]] = None + +class WebhookData(BaseModel): + """Data sent to webhooks.""" + job_id: str = Field(alias="jobId") + status: str + current: Optional[int] = None + total: Optional[int] = None + data: Optional[List[Document]] = None + error: Optional[str] = None + class Source(BaseModel): """Configuration for a search source.""" type: str SourceOption = Union[str, Source] - FormatString = Literal[ # camelCase versions (API format) "markdown", "html", "rawHtml", "links", "screenshot", "json", "changeTracking", @@ -183,7 +199,7 @@ class CrawlRequest(BaseModel): allow_subdomains: bool = False delay: Optional[int] = None max_concurrency: Optional[int] = None - webhook: Optional[Dict[str, Any]] = None + webhook: Optional[Union[str, WebhookConfig]] = None scrape_options: Optional[ScrapeOptions] = None zero_data_retention: bool = False @@ -231,7 +247,7 @@ class CrawlParamsData(BaseModel): allow_subdomains: bool = False delay: Optional[int] = None max_concurrency: Optional[int] = None - webhook: Optional[Dict[str, Any]] = None + webhook: Optional[Union[str, WebhookConfig]] = None scrape_options: Optional[ScrapeOptions] = None zero_data_retention: bool = False warning: Optional[str] = None @@ -343,12 +359,6 @@ class Location(BaseModel): country: Optional[str] = None languages: Optional[List[str]] = None -# JSON format types -class JsonFormat(BaseModel): - """Configuration for JSON extraction.""" - prompt: Optional[str] = None - schema_field: Optional[Dict[str, Any]] = Field(None, alias="schema") - class SearchRequest(BaseModel): """Request for search operations.""" query: str @@ -420,15 +430,22 @@ class JobStatus(BaseModel): completed_at: Optional[datetime] = Field(None, alias="completedAt") expires_at: Optional[datetime] = Field(None, alias="expiresAt") -# Webhook types -class WebhookData(BaseModel): - """Data sent to webhooks.""" - job_id: str = Field(alias="jobId") - status: str - current: Optional[int] = None - total: Optional[int] = None - data: Optional[List[Document]] = None - error: Optional[str] = None +class CrawlErrorsResponse(BaseModel): + """Response from crawl error monitoring.""" + errors: List[Dict[str, str]] = Field(description="List of errors with fields: id, timestamp, url, error") + robots_blocked: List[str] = Field(alias="robotsBlocked", description="List of URLs blocked by robots.txt") + +class ActiveCrawl(BaseModel): + """Information about an active crawl job.""" + id: str + team_id: str = Field(alias="teamId") + url: str + options: Optional[Dict[str, Any]] = None + +class ActiveCrawlsResponse(BaseModel): + """Response from active crawls endpoint.""" + success: bool = True + crawls: List[ActiveCrawl] # Configuration types class ClientConfig(BaseModel): @@ -446,5 +463,5 @@ AnyResponse = Union[ BatchScrapeResponse, MapResponse, SearchResponse, - ErrorResponse + ErrorResponse, ] \ No newline at end of file