Package scrapfly
Sub-modules
scrapfly.api_configscrapfly.api_responsescrapfly.batch-
Streaming multipart/mixed parser for the POST /scrape/batch endpoint …
scrapfly.browser_configscrapfly.classify-
Classify API response model …
scrapfly.clientscrapfly.crawler-
Scrapfly Crawler API …
scrapfly.errorsscrapfly.extraction_configscrapfly.frozen_dictscrapfly.polyfillscrapfly.reporterscrapfly.schedule-
Public schedule client for the Scrapfly API …
scrapfly.scrape_configscrapfly.scrapyscrapfly.screenshot_configscrapfly.webhook
Functions
def parse_warc(warc_data: bytes |) ‑> WarcParser -
Expand source code
def parse_warc(warc_data: Union[bytes, BinaryIO]) -> WarcParser: """ Convenience function to create a WARC parser Args: warc_data: WARC data as bytes or file-like object Returns: WarcParser: Parser instance Example: ```python from scrapfly import parse_warc # Quick way to get all pages pages = parse_warc(warc_bytes).get_pages() for page in pages: print(f"{page['url']}: {page['status_code']}") ``` """ return WarcParser(warc_data)Convenience function to create a WARC parser
Args
warc_data- WARC data as bytes or file-like object
Returns
WarcParser- Parser instance
Example
from scrapfly import parse_warc # Quick way to get all pages pages = parse_warc(warc_bytes).get_pages() for page in pages: print(f"{page['url']}: {page['status_code']}") def webhook_from_payload(payload: Dict[str, Any] | None = None,
signing_secrets: Tuple[str, ...] | None = None,
signature: str | None = None,
raw_body: bytes | None = None,
content_encoding: str | None = None) ‑> CrawlerLifecycleWebhook | CrawlerUrlVisitedWebhook | CrawlerUrlSkippedWebhook | CrawlerUrlDiscoveredWebhook | CrawlerUrlFailedWebhook | CrawlerSearchWebhook | CrawlerUpdatedWebhook-
Expand source code
def webhook_from_payload( payload: Optional[Dict[str, Any]] = None, signing_secrets: Optional[Tuple[str, ...]] = None, signature: Optional[str] = None, raw_body: Optional[bytes] = None, content_encoding: Optional[str] = None, ) -> CrawlerWebhook: """ Parse a raw crawler webhook envelope into a typed dataclass. The envelope shape is ``{"event": <name>, "payload": {...}}``. This function inspects ``event`` and returns the corresponding typed dataclass — one of :data:`CrawlerWebhook`. Args: payload: The full webhook body as a dict (i.e. what you get from ``request.json``). Ignored when ``signing_secrets`` is set, because the envelope is then re-read from the verified bytes instead — returning an object built from an unverified dict would make the verification decorative. signing_secrets: Optional tuple of signing secrets for signature verification. Pass each secret as it appears in the webhook dashboard (UTF-8 string, not hex-encoded). signature: Optional webhook signature header value (``X-Scrapfly-Webhook-Signature``). raw_body: The exact request bytes (``request.get_data()``). Required when ``signing_secrets`` is set: the signature covers the bytes on the wire, and re-serializing the parsed dict does not reproduce them (separators, float repr, unicode escaping and key order are all encoder-dependent). content_encoding: The ``Content-Encoding`` header, when the webhook is configured to compress. Signing happens before encoding, so a compressed body has to be inflated before the digest matches. Returns: A typed webhook instance matching the event. Raises: KeyError: If the envelope is missing required fields. ValueError: If ``event`` is not one of the known crawler events, if ``signing_secrets`` is set without ``raw_body``, or if neither ``payload`` nor ``signing_secrets`` is supplied. WebhookSignatureMissMatch: If the signature is absent or does not match. Example: >>> from flask import Flask, request >>> from scrapfly import webhook_from_payload, CrawlerLifecycleWebhook >>> app = Flask(__name__) >>> @app.route('/webhook', methods=['POST']) ... def handle_webhook(): ... wh = webhook_from_payload( ... request.json, ... signing_secrets=('YOUR-WEBHOOK-SIGNING-SECRET',), ... signature=request.headers.get('X-Scrapfly-Webhook-Signature'), ... raw_body=request.get_data(), ... content_encoding=request.headers.get('Content-Encoding'), ... ) ... if isinstance(wh, CrawlerLifecycleWebhook) and wh.event == 'crawler_finished': ... print(f"Crawl {wh.crawler_uuid} finished — " ... f"{wh.state.urls_visited} URLs visited") ... return '', 200 """ if signing_secrets: # Imported here rather than at module scope to avoid a circular import. from json import loads from ..api_response import ResponseBodyHandler, decompress from ..errors import WebhookSignatureMissMatch if raw_body is None: raise ValueError( "signature verification requires raw_body (the exact request bytes); " "the parsed payload cannot reproduce the signed message" ) handler = ResponseBodyHandler(signing_secrets=signing_secrets) # Signing happens before Content-Encoding is applied. verified = decompress(raw_body, content_encoding) if not handler.verify(verified, signature): raise WebhookSignatureMissMatch() # Parse what was actually signed. Building the result from the caller's # dict would let unsigned fields ride in behind a valid signature. payload = loads(verified) elif payload is None: raise ValueError('webhook_from_payload needs either payload or signing_secrets + raw_body') event = payload['event'] inner = payload['payload'] parser = _DISPATCH.get(event) if parser is None: raise ValueError( f"Unknown crawler webhook event: {event!r}. " f"Expected one of: {sorted(_DISPATCH.keys())}" ) return parser.from_payload(event, inner)Parse a raw crawler webhook envelope into a typed dataclass.
The envelope shape is
{"event": <name>, "payload": {...}}. This function inspectseventand returns the corresponding typed dataclass — one of :data:CrawlerWebhook.Args
payload- The full webhook body as a dict (i.e. what you get from
request.json). Ignored whensigning_secretsis set, because the envelope is then re-read from the verified bytes instead — returning an object built from an unverified dict would make the verification decorative. signing_secrets- Optional tuple of signing secrets for signature verification. Pass each secret as it appears in the webhook dashboard (UTF-8 string, not hex-encoded).
signature- Optional webhook signature header value
(
X-Scrapfly-Webhook-Signature). raw_body- The exact request bytes (
request.get_data()). Required whensigning_secretsis set: the signature covers the bytes on the wire, and re-serializing the parsed dict does not reproduce them (separators, float repr, unicode escaping and key order are all encoder-dependent). content_encoding- The
Content-Encodingheader, when the webhook is configured to compress. Signing happens before encoding, so a compressed body has to be inflated before the digest matches.
Returns
A typed webhook instance matching the event.
Raises
KeyError- If the envelope is missing required fields.
ValueError- If
eventis not one of the known crawler events, ifsigning_secretsis set withoutraw_body, or if neitherpayloadnorsigning_secretsis supplied. WebhookSignatureMissMatch- If the signature is absent or does not match.
Example
>>> from flask import Flask, request >>> from scrapfly import webhook_from_payload, CrawlerLifecycleWebhook >>> app = Flask(__name__) >>> @app.route('/webhook', methods=['POST']) ... def handle_webhook(): ... wh = webhook_from_payload( ... request.json, ... signing_secrets=('YOUR-WEBHOOK-SIGNING-SECRET',), ... signature=request.headers.get('X-Scrapfly-Webhook-Signature'), ... raw_body=request.get_data(), ... content_encoding=request.headers.get('Content-Encoding'), ... ) ... if isinstance(wh, CrawlerLifecycleWebhook) and wh.event == 'crawler_finished': ... print(f"Crawl {wh.crawler_uuid} finished — " ... f"{wh.state.urls_visited} URLs visited") ... return '', 200
Classes
class ApiHttpClientError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class ApiHttpClientError(HttpError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
Subclasses
- ApiHttpServerError
- scrapfly.errors.BadApiKeyError
- scrapfly.errors.PaymentRequired
- scrapfly.errors.TooManyRequest
class ApiHttpServerError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class ApiHttpServerError(ApiHttpClientError): passCommon base class for all non-exit exceptions.
Ancestors
- ApiHttpClientError
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class BrowserConfig (proxy_pool: str | ProxyPool | None = None,
os: str | OperatingSystem | None = None,
session: str | None = None,
country: str | None = None,
lang: str | None = None,
languages: str | List[str] | None = None,
auto_close: bool | None = None,
timeout: int | None = None,
debug: bool | None = None,
extensions: List[str] | None = None,
block_images: bool | None = None,
block_styles: bool | None = None,
block_fonts: bool | None = None,
block_media: bool | None = None,
screenshot: bool | None = None,
resolution: str | None = None,
target_url: str | None = None,
cache: bool | None = None,
blacklist: bool | None = None,
unblock: bool | None = None,
unblock_timeout: int | None = None,
browser_brand: str | None = None,
byop_proxy: str | None = None,
enable_mcp: bool | None = None,
solve_captcha: bool | None = None,
vault: str | None = None,
vault_key: str | None = None,
enable_vnc: bool | None = None,
vnc_password: str | None = None,
enable_rtc: bool | None = None,
rtc_username: str | None = None,
rtc_password: str | None = None,
hitl_allowed_networks: str | List[str] | None = None)-
Expand source code
class BrowserConfig(BaseApiConfig): CLOUD_BROWSER_HOST = 'wss://browser.scrapfly.io' def __init__( self, proxy_pool: Optional[Union[str, ProxyPool]] = None, os: Optional[Union[str, OperatingSystem]] = None, session: Optional[str] = None, country: Optional[str] = None, lang: Optional[str] = None, languages: Optional[Union[str, List[str]]] = None, auto_close: Optional[bool] = None, timeout: Optional[int] = None, debug: Optional[bool] = None, extensions: Optional[List[str]] = None, block_images: Optional[bool] = None, block_styles: Optional[bool] = None, block_fonts: Optional[bool] = None, block_media: Optional[bool] = None, screenshot: Optional[bool] = None, resolution: Optional[str] = None, target_url: Optional[str] = None, cache: Optional[bool] = None, blacklist: Optional[bool] = None, unblock: Optional[bool] = None, unblock_timeout: Optional[int] = None, browser_brand: Optional[str] = None, byop_proxy: Optional[str] = None, enable_mcp: Optional[bool] = None, solve_captcha: Optional[bool] = None, vault: Optional[str] = None, vault_key: Optional[str] = None, enable_vnc: Optional[bool] = None, vnc_password: Optional[str] = None, enable_rtc: Optional[bool] = None, rtc_username: Optional[str] = None, rtc_password: Optional[str] = None, hitl_allowed_networks: Optional[Union[str, List[str]]] = None, ): if timeout is not None and timeout > 1800: raise ValueError('timeout cannot exceed 1800 seconds (30 minutes)') if isinstance(proxy_pool, str): try: proxy_pool = ProxyPool(proxy_pool) except ValueError: pass if isinstance(os, str): try: os = OperatingSystem(os) except ValueError: pass self.proxy_pool = proxy_pool self.os = os self.session = session self.country = country # Browser UI language — the singular navigator.language base tag # (e.g. "en"). When omitted, the server derives it from `country`. self.lang = lang # Ordered language preference list — drives navigator.languages and # the q-weighted Accept-Language header (e.g. ["fr-FR", "fr", "en-US"]). # Capped server-side at 3 entries; accept a list or a comma-separated # string and normalize to a comma-separated string on the wire (the # same shape `extensions` uses), so the API can split it back. if isinstance(languages, list): languages = ','.join(s.strip() for s in languages if s and s.strip()) self.languages = languages or None self.auto_close = auto_close self.timeout = timeout self.debug = debug self.extensions = extensions self.block_images = block_images self.block_styles = block_styles self.block_fonts = block_fonts self.block_media = block_media self.screenshot = screenshot self.resolution = resolution self.target_url = target_url self.cache = cache self.blacklist = blacklist self.unblock = unblock self.unblock_timeout = unblock_timeout self.browser_brand = browser_brand # BYOP (Bring Your Own Proxy): full proxy URL # Format: {protocol}://{user}:{pass}@{host}:{port} # Supported protocols: http, https, socks5, socks5h, socks5+udp, socks5h+udp # The +udp variants enable HTTP/3 (QUIC) via SOCKS5 UDP ASSOCIATE — only # works with proxy providers that implement RFC 1928 UDP ASSOCIATE. # Requires a Custom plan subscription. See: # https://scrapfly.io/docs/cloud-browser-api/byop self.byop_proxy = byop_proxy self.enable_mcp = enable_mcp # SolveCaptcha: arm the Cloud Browser's built-in captcha detector + solver on # the first page attach. Turnstile, DataDome slider, reCAPTCHA, # GeeTest, PerimeterX hold, and puzzle captchas are handled # automatically. Billed per solve; failures cost nothing. # https://scrapfly.io/docs/cloud-browser-api/captcha-solver self.solve_captcha = solve_captcha # Cloud Browser Credential Vault binding. `vault` is the vault NAME # (the human name you gave it at POST /vault — alphanumeric); the # server resolves it to the underlying vault scoped to your api-key's # project and environment. `vault_key` is the base64 32-byte # customer-held key returned (once) by POST /vault or POST # /vault/<id>/rotate. Both are forwarded as query params on the # wss:// URL — the server uses them to decrypt items into the # session's secret broker. Treat `vault_key` as a credential: never # log it, never persist it server-side. self.vault = vault self.vault_key = vault_key self.enable_vnc = enable_vnc self.vnc_password = vnc_password self.enable_rtc = enable_rtc self.rtc_username = rtc_username self.rtc_password = rtc_password if isinstance(hitl_allowed_networks, list): hitl_allowed_networks = ','.join(s.strip() for s in hitl_allowed_networks if s and s.strip()) self.hitl_allowed_networks = hitl_allowed_networks or None @staticmethod def project_salt(api_key: str) -> str: """Return the deterministic project salt for an api_key: sha256(api_key)[:8]. The server returns the same value via the X-Browser-Project-Salt response header on VNC-enabled upgrades, where the salt is also the VNC password prefix (<salt>-<password>). """ return hashlib.sha256(api_key.encode('utf-8')).hexdigest()[:8] def vnc_client_password(self, api_key: str) -> str: """Return the password a native VNC client must type to attach to a session created with this config: "<project_salt>-<vnc_password>". Required by the VNC TCP endpoint (port 5901), which the server salts at allocation. The WebSocket endpoint /run/<run_id>/vnc takes the raw vnc_password instead. """ if not self.enable_vnc or not self.vnc_password: raise ValueError('enable_vnc and vnc_password must both be set on this BrowserConfig') return f'{self.project_salt(api_key)}-{self.vnc_password}' def websocket_url(self, api_key: str, host: Optional[str] = None) -> str: params = {'api_key': api_key} if self.proxy_pool is not None: params['proxy_pool'] = self.proxy_pool.value if isinstance(self.proxy_pool, ProxyPool) else self.proxy_pool if self.os is not None: params['os'] = self.os.value if isinstance(self.os, OperatingSystem) else self.os if self.session is not None: params['session'] = self.session if self.country is not None: params['country'] = self.country if self.lang is not None: params['lang'] = self.lang if self.languages is not None: params['languages'] = self.languages if self.auto_close is not None: params['auto_close'] = self._bool_to_http(self.auto_close) if self.timeout is not None: params['timeout'] = str(self.timeout) if self.debug is not None: params['debug'] = self._bool_to_http(self.debug) if self.extensions: params['extensions'] = ','.join(self.extensions) if self.block_images is not None: params['block_images'] = self._bool_to_http(self.block_images) if self.block_styles is not None: params['block_styles'] = self._bool_to_http(self.block_styles) if self.block_fonts is not None: params['block_fonts'] = self._bool_to_http(self.block_fonts) if self.block_media is not None: params['block_media'] = self._bool_to_http(self.block_media) if self.screenshot is not None: params['screenshot'] = self._bool_to_http(self.screenshot) if self.resolution is not None: params['resolution'] = self.resolution if self.target_url is not None: params['target_url'] = self.target_url if self.cache is not None: params['cache'] = self._bool_to_http(self.cache) if self.blacklist is not None: params['blacklist'] = self._bool_to_http(self.blacklist) if self.unblock is not None: params['unblock'] = self._bool_to_http(self.unblock) if self.unblock_timeout is not None: params['unblock_timeout'] = str(self.unblock_timeout) if self.browser_brand is not None: params['browser_brand'] = self.browser_brand if self.byop_proxy is not None: params['byop_proxy'] = self.byop_proxy if self.enable_mcp is not None: params['enable_mcp'] = self._bool_to_http(self.enable_mcp) if self.solve_captcha is not None: params['solve_captcha'] = self._bool_to_http(self.solve_captcha) if self.vault is not None: params['vault'] = self.vault if self.vault_key is not None: params['vault_key'] = self.vault_key if self.enable_vnc is not None: params['enable_vnc'] = self._bool_to_http(self.enable_vnc) if self.vnc_password is not None: params['vnc_password'] = self.vnc_password if self.enable_rtc is not None: params['enable_rtc'] = self._bool_to_http(self.enable_rtc) if self.rtc_username is not None: params['rtc_username'] = self.rtc_username if self.rtc_password is not None: params['rtc_password'] = self.rtc_password if self.hitl_allowed_networks is not None: params['hitl_allowed_networks'] = self.hitl_allowed_networks base_host = host or self.CLOUD_BROWSER_HOST return base_host + '?' + urlencode(params) def to_dict(self) -> Dict: return { 'proxy_pool': self.proxy_pool.value if isinstance(self.proxy_pool, ProxyPool) else self.proxy_pool, 'os': self.os.value if isinstance(self.os, OperatingSystem) else self.os, 'session': self.session, 'country': self.country, 'lang': self.lang, 'languages': self.languages, 'auto_close': self.auto_close, 'timeout': self.timeout, 'debug': self.debug, 'extensions': self.extensions, 'block_images': self.block_images, 'block_styles': self.block_styles, 'block_fonts': self.block_fonts, 'block_media': self.block_media, 'screenshot': self.screenshot, 'resolution': self.resolution, 'target_url': self.target_url, 'cache': self.cache, 'blacklist': self.blacklist, 'unblock': self.unblock, 'unblock_timeout': self.unblock_timeout, 'browser_brand': self.browser_brand, 'byop_proxy': self.byop_proxy, 'enable_mcp': self.enable_mcp, 'solve_captcha': self.solve_captcha, 'vault': self.vault, 'vault_key': self.vault_key, 'enable_vnc': self.enable_vnc, 'vnc_password': self.vnc_password, 'enable_rtc': self.enable_rtc, 'rtc_username': self.rtc_username, 'rtc_password': self.rtc_password, 'hitl_allowed_networks': self.hitl_allowed_networks, } @staticmethod def from_dict(browser_config_dict: Dict) -> 'BrowserConfig': return BrowserConfig( proxy_pool=browser_config_dict.get('proxy_pool', None), os=browser_config_dict.get('os', None), session=browser_config_dict.get('session', None), country=browser_config_dict.get('country', None), lang=browser_config_dict.get('lang', None), languages=browser_config_dict.get('languages', None), auto_close=browser_config_dict.get('auto_close', None), timeout=browser_config_dict.get('timeout', None), debug=browser_config_dict.get('debug', None), extensions=browser_config_dict.get('extensions', None), block_images=browser_config_dict.get('block_images', None), block_styles=browser_config_dict.get('block_styles', None), block_fonts=browser_config_dict.get('block_fonts', None), block_media=browser_config_dict.get('block_media', None), screenshot=browser_config_dict.get('screenshot', None), resolution=browser_config_dict.get('resolution', None), target_url=browser_config_dict.get('target_url', None), cache=browser_config_dict.get('cache', None), blacklist=browser_config_dict.get('blacklist', None), unblock=browser_config_dict.get('unblock', None), unblock_timeout=browser_config_dict.get('unblock_timeout', None), browser_brand=browser_config_dict.get('browser_brand', None), byop_proxy=browser_config_dict.get('byop_proxy', None), enable_mcp=browser_config_dict.get('enable_mcp', None), solve_captcha=browser_config_dict.get('solve_captcha', None), vault=browser_config_dict.get('vault', None), vault_key=browser_config_dict.get('vault_key', None), enable_vnc=browser_config_dict.get('enable_vnc', None), vnc_password=browser_config_dict.get('vnc_password', None), enable_rtc=browser_config_dict.get('enable_rtc', None), rtc_username=browser_config_dict.get('rtc_username', None), rtc_password=browser_config_dict.get('rtc_password', None), hitl_allowed_networks=browser_config_dict.get('hitl_allowed_networks', None), )Ancestors
Class variables
var CLOUD_BROWSER_HOST
Static methods
def from_dict(browser_config_dict: Dict) ‑> BrowserConfig-
Expand source code
@staticmethod def from_dict(browser_config_dict: Dict) -> 'BrowserConfig': return BrowserConfig( proxy_pool=browser_config_dict.get('proxy_pool', None), os=browser_config_dict.get('os', None), session=browser_config_dict.get('session', None), country=browser_config_dict.get('country', None), lang=browser_config_dict.get('lang', None), languages=browser_config_dict.get('languages', None), auto_close=browser_config_dict.get('auto_close', None), timeout=browser_config_dict.get('timeout', None), debug=browser_config_dict.get('debug', None), extensions=browser_config_dict.get('extensions', None), block_images=browser_config_dict.get('block_images', None), block_styles=browser_config_dict.get('block_styles', None), block_fonts=browser_config_dict.get('block_fonts', None), block_media=browser_config_dict.get('block_media', None), screenshot=browser_config_dict.get('screenshot', None), resolution=browser_config_dict.get('resolution', None), target_url=browser_config_dict.get('target_url', None), cache=browser_config_dict.get('cache', None), blacklist=browser_config_dict.get('blacklist', None), unblock=browser_config_dict.get('unblock', None), unblock_timeout=browser_config_dict.get('unblock_timeout', None), browser_brand=browser_config_dict.get('browser_brand', None), byop_proxy=browser_config_dict.get('byop_proxy', None), enable_mcp=browser_config_dict.get('enable_mcp', None), solve_captcha=browser_config_dict.get('solve_captcha', None), vault=browser_config_dict.get('vault', None), vault_key=browser_config_dict.get('vault_key', None), enable_vnc=browser_config_dict.get('enable_vnc', None), vnc_password=browser_config_dict.get('vnc_password', None), enable_rtc=browser_config_dict.get('enable_rtc', None), rtc_username=browser_config_dict.get('rtc_username', None), rtc_password=browser_config_dict.get('rtc_password', None), hitl_allowed_networks=browser_config_dict.get('hitl_allowed_networks', None), ) def project_salt(api_key: str) ‑> str-
Expand source code
@staticmethod def project_salt(api_key: str) -> str: """Return the deterministic project salt for an api_key: sha256(api_key)[:8]. The server returns the same value via the X-Browser-Project-Salt response header on VNC-enabled upgrades, where the salt is also the VNC password prefix (<salt>-<password>). """ return hashlib.sha256(api_key.encode('utf-8')).hexdigest()[:8]Return the deterministic project salt for an api_key: sha256(api_key)[:8]. The server returns the same value via the X-Browser-Project-Salt response header on VNC-enabled upgrades, where the salt is also the VNC password prefix (
- ).
Methods
def to_dict(self) ‑> Dict-
Expand source code
def to_dict(self) -> Dict: return { 'proxy_pool': self.proxy_pool.value if isinstance(self.proxy_pool, ProxyPool) else self.proxy_pool, 'os': self.os.value if isinstance(self.os, OperatingSystem) else self.os, 'session': self.session, 'country': self.country, 'lang': self.lang, 'languages': self.languages, 'auto_close': self.auto_close, 'timeout': self.timeout, 'debug': self.debug, 'extensions': self.extensions, 'block_images': self.block_images, 'block_styles': self.block_styles, 'block_fonts': self.block_fonts, 'block_media': self.block_media, 'screenshot': self.screenshot, 'resolution': self.resolution, 'target_url': self.target_url, 'cache': self.cache, 'blacklist': self.blacklist, 'unblock': self.unblock, 'unblock_timeout': self.unblock_timeout, 'browser_brand': self.browser_brand, 'byop_proxy': self.byop_proxy, 'enable_mcp': self.enable_mcp, 'solve_captcha': self.solve_captcha, 'vault': self.vault, 'vault_key': self.vault_key, 'enable_vnc': self.enable_vnc, 'vnc_password': self.vnc_password, 'enable_rtc': self.enable_rtc, 'rtc_username': self.rtc_username, 'rtc_password': self.rtc_password, 'hitl_allowed_networks': self.hitl_allowed_networks, } def vnc_client_password(self, api_key: str) ‑> str-
Expand source code
def vnc_client_password(self, api_key: str) -> str: """Return the password a native VNC client must type to attach to a session created with this config: "<project_salt>-<vnc_password>". Required by the VNC TCP endpoint (port 5901), which the server salts at allocation. The WebSocket endpoint /run/<run_id>/vnc takes the raw vnc_password instead. """ if not self.enable_vnc or not self.vnc_password: raise ValueError('enable_vnc and vnc_password must both be set on this BrowserConfig') return f'{self.project_salt(api_key)}-{self.vnc_password}'Return the password a native VNC client must type to attach to a session created with this config: "
- ". Required by the VNC TCP endpoint (port 5901), which the server salts at allocation. The WebSocket endpoint /run/
/vnc takes the raw vnc_password instead. def websocket_url(self, api_key: str, host: str | None = None) ‑> str-
Expand source code
def websocket_url(self, api_key: str, host: Optional[str] = None) -> str: params = {'api_key': api_key} if self.proxy_pool is not None: params['proxy_pool'] = self.proxy_pool.value if isinstance(self.proxy_pool, ProxyPool) else self.proxy_pool if self.os is not None: params['os'] = self.os.value if isinstance(self.os, OperatingSystem) else self.os if self.session is not None: params['session'] = self.session if self.country is not None: params['country'] = self.country if self.lang is not None: params['lang'] = self.lang if self.languages is not None: params['languages'] = self.languages if self.auto_close is not None: params['auto_close'] = self._bool_to_http(self.auto_close) if self.timeout is not None: params['timeout'] = str(self.timeout) if self.debug is not None: params['debug'] = self._bool_to_http(self.debug) if self.extensions: params['extensions'] = ','.join(self.extensions) if self.block_images is not None: params['block_images'] = self._bool_to_http(self.block_images) if self.block_styles is not None: params['block_styles'] = self._bool_to_http(self.block_styles) if self.block_fonts is not None: params['block_fonts'] = self._bool_to_http(self.block_fonts) if self.block_media is not None: params['block_media'] = self._bool_to_http(self.block_media) if self.screenshot is not None: params['screenshot'] = self._bool_to_http(self.screenshot) if self.resolution is not None: params['resolution'] = self.resolution if self.target_url is not None: params['target_url'] = self.target_url if self.cache is not None: params['cache'] = self._bool_to_http(self.cache) if self.blacklist is not None: params['blacklist'] = self._bool_to_http(self.blacklist) if self.unblock is not None: params['unblock'] = self._bool_to_http(self.unblock) if self.unblock_timeout is not None: params['unblock_timeout'] = str(self.unblock_timeout) if self.browser_brand is not None: params['browser_brand'] = self.browser_brand if self.byop_proxy is not None: params['byop_proxy'] = self.byop_proxy if self.enable_mcp is not None: params['enable_mcp'] = self._bool_to_http(self.enable_mcp) if self.solve_captcha is not None: params['solve_captcha'] = self._bool_to_http(self.solve_captcha) if self.vault is not None: params['vault'] = self.vault if self.vault_key is not None: params['vault_key'] = self.vault_key if self.enable_vnc is not None: params['enable_vnc'] = self._bool_to_http(self.enable_vnc) if self.vnc_password is not None: params['vnc_password'] = self.vnc_password if self.enable_rtc is not None: params['enable_rtc'] = self._bool_to_http(self.enable_rtc) if self.rtc_username is not None: params['rtc_username'] = self.rtc_username if self.rtc_password is not None: params['rtc_password'] = self.rtc_password if self.hitl_allowed_networks is not None: params['hitl_allowed_networks'] = self.hitl_allowed_networks base_host = host or self.CLOUD_BROWSER_HOST return base_host + '?' + urlencode(params)
class ClassifyResult (blocked: bool, antibot: Optional[str], cost: int)-
Expand source code
@dataclass class ClassifyResult: blocked: bool antibot: Optional[str] cost: int @classmethod def from_dict(cls, data: dict) -> "ClassifyResult": return cls( blocked=bool(data.get("blocked", False)), antibot=data.get("antibot"), cost=int(data.get("cost", 0)), )ClassifyResult(blocked: 'bool', antibot: 'Optional[str]', cost: 'int')
Static methods
def from_dict(data: dict) ‑> ClassifyResult
Instance variables
var antibot : str | Nonevar blocked : boolvar cost : int
class Crawl (client: ScrapflyClient,
config: CrawlerConfig)-
Expand source code
class Crawl: """ High-level abstraction for managing a crawler job The Crawl object maintains the state of a crawler job and provides convenient methods for managing its lifecycle. Example: ```python from scrapfly import ScrapflyClient, CrawlerConfig, Crawl client = ScrapflyClient(key='your-key') config = CrawlerConfig(url='https://example.com', page_limit=10) # Create and start crawl crawl = Crawl(client, config) crawl.crawl() # Start the crawler # Wait for completion crawl.wait() # Get results pages = crawl.warc().get_pages() for page in pages: print(f"{page['url']}: {page['status_code']}") # Or read specific URLs html = crawl.read('https://example.com/page1', format='html') ``` """ def __init__(self, client: 'ScrapflyClient', config: CrawlerConfig): """ Initialize a Crawl object Args: client: ScrapflyClient instance config: CrawlerConfig with crawler settings """ self._client = client self._config = config self._uuid: Optional[str] = None self._status_cache: Optional[CrawlerStatusResponse] = None self._artifact_cache: Optional[CrawlerArtifactResponse] = None @property def uuid(self) -> Optional[str]: """Get the crawler job UUID (None if not started)""" return self._uuid @property def started(self) -> bool: """Check if the crawler has been started""" return self._uuid is not None def crawl(self) -> 'Crawl': """ Start the crawler job Returns: Self for method chaining Raises: RuntimeError: If crawler already started Example: ```python crawl = Crawl(client, config) crawl.crawl() # Start crawling ``` """ if self._uuid is not None: raise ScrapflyCrawlerError( message="Crawler already started", code="ALREADY_STARTED", http_status_code=400 ) response = self._client.start_crawl(self._config) self._uuid = response.uuid return self def status(self, refresh: bool = True) -> CrawlerStatusResponse: """ Get current crawler status Args: refresh: If True, fetch fresh status from API. If False, return cached status. Returns: CrawlerStatusResponse with current status Raises: RuntimeError: If crawler not started yet Example: ```python status = crawl.status() print(f"Progress: {status.progress_pct}%") print(f"URLs visited: {status.state.urls_visited}") ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) if refresh or self._status_cache is None: self._status_cache = self._client.get_crawl_status(self._uuid) return self._status_cache def wait( self, poll_interval: int = 5, max_wait: Optional[int] = None, verbose: bool = False, allow_cancelled: bool = False, ) -> 'Crawl': """ Wait for crawler to complete Polls the status endpoint until the crawler finishes. Args: poll_interval: Seconds between status checks (default: 5) max_wait: Maximum seconds to wait (None = wait forever) verbose: If True, print progress updates allow_cancelled: If True, return normally when the crawler reaches CANCELLED instead of raising. Useful for the cancel-then-wait pattern where the caller already knows they triggered the cancellation. Defaults to False (raises ScrapflyCrawlerError with code='CANCELLED' on user_cancelled), preserving prior behavior for callers that observe external cancellations. Returns: Self for method chaining Raises: ScrapflyCrawlerError: If crawler not started, failed, or timed out. Also raised on cancellation when ``allow_cancelled=False``. Example: ```python # Wait with progress updates crawl.crawl().wait(verbose=True) # Wait with timeout crawl.crawl().wait(max_wait=300) # 5 minutes max # Cancel from the same call site, then wait without re-raising crawl.cancel() crawl.wait(allow_cancelled=True) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) start_time = time.time() poll_count = 0 while True: status = self.status(refresh=True) poll_count += 1 if verbose: logger.info(f"Poll #{poll_count}: {status.status} - " f"{status.progress_pct:.1f}% - " f"{status.state.urls_visited}/{status.state.urls_extracted} URLs") if status.is_complete: if verbose: logger.info(f"✓ Crawler completed successfully!") return self elif status.is_failed: raise ScrapflyCrawlerError( message=f"Crawler failed with status: {status.status}", code="FAILED", http_status_code=400 ) elif status.is_cancelled: if allow_cancelled: if verbose: logger.info("Crawler was cancelled (allow_cancelled=True)") return self raise ScrapflyCrawlerError( message="Crawler was cancelled", code="CANCELLED", http_status_code=400 ) # Check timeout if max_wait is not None: elapsed = time.time() - start_time if elapsed > max_wait: raise ScrapflyCrawlerError( message=f"Timeout waiting for crawler (>{max_wait}s)", code="TIMEOUT", http_status_code=400 ) time.sleep(poll_interval) def cancel(self) -> bool: """ Cancel the running crawler job Returns: True if cancelled successfully Raises: ScrapflyCrawlerError: If crawler not started yet Example: ```python # Start a crawl crawl = Crawl(client, config).crawl() # Cancel it crawl.cancel() ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) return self._client.cancel_crawl(self._uuid) def urls( self, status: Optional[Literal['visited', 'pending', 'failed', 'skipped']] = None, page: int = 1, per_page: int = 100, ) -> CrawlerUrlsResponse: """ List the crawled URLs, optionally filtered by status. Convenience wrapper around :meth:`ScrapflyClient.get_crawl_urls` that pre-fills the crawler UUID. Args: status: Filter by URL status: 'visited', 'pending', 'failed' or 'skipped'. When None, the server defaults to 'visited'. page: 1-based page number, echoed on the response. per_page: Page size, echoed on the response. Returns: CrawlerUrlsResponse with the URL records and the echoed pagination. Raises: ScrapflyCrawlerError: if the crawler has not been started yet. Example: ```python crawl = Crawl(client, config).crawl().wait() for entry in crawl.urls(status='visited'): print(f"{entry.url} ({entry.status})") ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400, ) return self._client.get_crawl_urls( uuid=self._uuid, status=status, page=page, per_page=per_page, ) def search( self, query: str, limit: int = 10, mode: Literal['vector', 'fts', 'hybrid'] = 'hybrid', filters: Optional[Dict[str, Any]] = None, cursor: Optional[str] = None, ) -> CrawlerSearchResponse: """ Search this crawl's index. Sugar over :meth:`ScrapflyClient.crawl_search` with a one-element crawl list. The cross-crawl call is the real endpoint; this shares its implementation so single and multi-crawl search cannot drift. Requires the crawl to have been started with ``CrawlerConfig(search=True)`` and its index to have reached ``READY``/``PARTIAL``; poll :attr:`CrawlerStatusResponse.search` or subscribe to the ``crawler_search_ready`` webhook. An index that is not ready yet comes back in ``response.skipped``, not as an error. Args: query: Free-text query. limit: Maximum results, 1-50 (server cap). mode: 'vector', 'fts' or 'hybrid'. filters: Optional flat filter map (see ``crawl_search``). cursor: Opaque next-page token from a previous response. Returns: CrawlerSearchResponse Raises: ScrapflyCrawlerError: if the crawler has not been started yet. Example: ```python crawl = Crawl(client, CrawlerConfig(url='https://example.com', search=True)).crawl().wait() for hit in crawl.search('pricing', limit=5): print(hit.rank, hit.url) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400, ) return self._client.crawl_search( crawl_ids=[self._uuid], query=query, limit=limit, mode=mode, filters=filters, cursor=cursor, ) def prompt( self, prompt: str, search: Optional[Dict[str, Any]] = None, model: Optional[str] = None, stream: bool = True, ) -> Union[Iterator[CrawlerPromptEvent], Dict[str, Any]]: """ Ask a question answered from this crawl's content. Sugar over :meth:`ScrapflyClient.crawl_prompt` with a one-element crawl list. Args: prompt: The question. search: Optional retrieval overrides: 'limit', 'mode', 'filters'. model: Optional Gemini model id; unset uses the server default. stream: Yield SSE frames (True) or return one dict (False). Returns: Iterator[CrawlerPromptEvent] when streaming, else Dict. Raises: ScrapflyCrawlerError: if the crawler has not been started yet. CrawlerPromptError: on a server-sent error frame, which can arrive after tokens have already been yielded. Example: ```python for event in crawl.prompt('What does this site sell?'): if event.is_token: print(event.data, end='', flush=True) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400, ) return self._client.crawl_prompt( crawl_ids=[self._uuid], prompt=prompt, search=search, model=model, stream=stream, ) def refresh_now(self) -> CrawlerRefreshState: """ Re-scrape this crawl's URLs in place, right now. Sugar over :meth:`ScrapflyClient.crawl_refresh_now`. The crawl keeps its uuid, its artifacts and its search index; only pages whose content changed are re-indexed and pages that disappeared are dropped, so anything already pointing at this crawl keeps working. Returns: CrawlerRefreshState Raises: ScrapflyCrawlerError: if the crawler has not been started yet. Example: ```python state = crawl.refresh_now() print(state.status) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400, ) return self._client.crawl_refresh_now(self._uuid) def refresh_settings( self, enabled: Optional[bool] = None, interval_seconds: Optional[int] = None, ) -> CrawlerRefreshState: """ Change this crawl's refresh schedule. Sugar over :meth:`ScrapflyClient.crawl_refresh_settings`. Only what is passed is changed. Args: enabled: Turn auto-refresh on or off. interval_seconds: Period between runs, 3600 to 7776000. Returns: CrawlerRefreshState Raises: ScrapflyCrawlerError: if the crawler has not been started yet. Example: ```python crawl.refresh_settings(enabled=True, interval_seconds=86400) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400, ) return self._client.crawl_refresh_settings( self._uuid, enabled=enabled, interval_seconds=interval_seconds, ) def refresh_history(self, limit: Optional[int] = None) -> List[CrawlerRefreshEntry]: """ Read this crawl's refresh timeline, newest last. Sugar over :meth:`ScrapflyClient.crawl_refresh_history`. Args: limit: Keep only the last N rows. Returns: List[CrawlerRefreshEntry] Raises: ScrapflyCrawlerError: if the crawler has not been started yet. Example: ```python for entry in crawl.refresh_history(limit=5): print(entry.at, entry.added, entry.updated, entry.removed) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400, ) return self._client.crawl_refresh_history(self._uuid, limit=limit) def warc(self, artifact_type: str = 'warc') -> CrawlerArtifactResponse: """ Download the crawler artifact (WARC file) Args: artifact_type: Type of artifact to download (default: 'warc') Returns: CrawlerArtifactResponse with parsed WARC data Raises: RuntimeError: If crawler not started yet Example: ```python # Get WARC artifact artifact = crawl.warc() # Get all pages pages = artifact.get_pages() # Iterate through responses for record in artifact.iter_responses(): print(record.url) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) if self._artifact_cache is None: self._artifact_cache = self._client.get_crawl_artifact( self._uuid, artifact_type=artifact_type ) return self._artifact_cache def har(self) -> CrawlerArtifactResponse: """ Download the crawler artifact in HAR (HTTP Archive) format Returns: CrawlerArtifactResponse with parsed HAR data Raises: RuntimeError: If crawler not started yet Example: ```python # Get HAR artifact artifact = crawl.har() # Get all pages pages = artifact.get_pages() # Iterate through HAR entries for entry in artifact.iter_responses(): print(f"{entry.url}: {entry.status_code}") print(f"Timing: {entry.time}ms") ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) return self._client.get_crawl_artifact( self._uuid, artifact_type='har' ) def read(self, url: str, format: ContentFormat = 'html') -> Optional[CrawlContent]: """ Read content from a specific URL in the crawl results Args: url: The URL to retrieve content for format: Content format - 'html', 'markdown', 'text', 'clean_html', 'json', 'extracted_data', 'page_metadata' Returns: CrawlContent object with content and metadata, or None if URL not found Example: ```python # Get HTML content for a specific URL content = crawl.read('https://example.com/page1') if content: print(f"URL: {content.url}") print(f"Status: {content.status_code}") print(f"Duration: {content.duration}s") print(content.content) # Get markdown content content = crawl.read('https://example.com/page1', format='markdown') if content: print(content.content) # Check if URL was crawled if crawl.read('https://example.com/missing') is None: print("URL not found in crawl results") ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) # For HTML format, we can get it from the WARC artifact (faster) if format == 'html': artifact = self.warc() for record in artifact.iter_responses(): if record.url == url: # Extract metadata from WARC headers warc_headers = record.warc_headers or {} duration_str = warc_headers.get('WARC-Scrape-Duration') duration = float(duration_str) if duration_str else None return CrawlContent( url=record.url, content=record.content.decode('utf-8', errors='replace'), status_code=record.status_code, headers=record.headers, duration=duration, log_id=warc_headers.get('WARC-Scrape-Log-Id'), country=warc_headers.get('WARC-Scrape-Country'), crawl_uuid=self._uuid ) return None # For other formats (markdown, text, etc.), use the contents API try: result = self._client.get_crawl_contents( self._uuid, format=format ) # The API returns: {"contents": {url: {format: content, ...}, ...}, "links": {...}} contents = result.get('contents', {}) if url in contents: content_data = contents[url] # Content is always a dict with format keys (e.g., {"html": "...", "markdown": "..."}) content_str = content_data.get(format) if content_str: # For non-HTML formats from contents API, we don't have full metadata # Try to get status code from WARC if possible status_code = 200 # Default headers = {} duration = None log_id = None country = None # Try to get metadata from WARC try: artifact = self.warc() for record in artifact.iter_responses(): if record.url == url: status_code = record.status_code headers = record.headers warc_headers = record.warc_headers or {} duration_str = warc_headers.get('WARC-Scrape-Duration') duration = float(duration_str) if duration_str else None log_id = warc_headers.get('WARC-Scrape-Log-Id') country = warc_headers.get('WARC-Scrape-Country') break except: pass return CrawlContent( url=url, content=content_str, status_code=status_code, headers=headers, duration=duration, log_id=log_id, country=country, crawl_uuid=self._uuid ) return None except Exception: # If contents API fails, return None return None def read_iter( self, pattern: str, format: ContentFormat = 'html' ) -> Iterator[CrawlContent]: """ Iterate through URLs matching a pattern and yield their content Supports wildcard patterns using * and ? for flexible URL matching. Args: pattern: URL pattern with wildcards (* matches any characters, ? matches one) Examples: "/products?page=*", "https://example.com/*/detail", "*/product/*" format: Content format to retrieve Yields: CrawlContent objects for each matching URL Example: ```python # Get all product pages in markdown for content in crawl.read_iter(pattern="*/products?page=*", format="markdown"): print(f"{content.url}: {len(content.content)} chars") print(f"Duration: {content.duration}s") # Get all detail pages for content in crawl.read_iter(pattern="*/detail/*"): process(content.content) # Pattern matching examples: # "/products?page=*" matches /products?page=1, /products?page=2, etc. # "*/product/*" matches any URL with /product/ in the path # "https://example.com/page?" matches https://example.com/page1, page2, etc. ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) # For HTML format, use WARC artifact (faster) if format == 'html': artifact = self.warc() for record in artifact.iter_responses(): if fnmatch.fnmatch(record.url, pattern): # Extract metadata from WARC headers warc_headers = record.warc_headers or {} duration_str = warc_headers.get('WARC-Scrape-Duration') duration = float(duration_str) if duration_str else None yield CrawlContent( url=record.url, content=record.content.decode('utf-8', errors='replace'), status_code=record.status_code, headers=record.headers, duration=duration, log_id=warc_headers.get('WARC-Scrape-Log-Id'), country=warc_headers.get('WARC-Scrape-Country'), crawl_uuid=self._uuid ) else: # For other formats, use contents API try: result = self._client.get_crawl_contents( self._uuid, format=format ) contents = result.get('contents', {}) # Build a metadata cache from WARC for non-HTML formats metadata_cache = {} try: artifact = self.warc() for record in artifact.iter_responses(): warc_headers = record.warc_headers or {} duration_str = warc_headers.get('WARC-Scrape-Duration') metadata_cache[record.url] = { 'status_code': record.status_code, 'headers': record.headers, 'duration': float(duration_str) if duration_str else None, 'log_id': warc_headers.get('WARC-Scrape-Log-Id'), 'country': warc_headers.get('WARC-Scrape-Country') } except: pass # Iterate through matching URLs for url, content_data in contents.items(): if fnmatch.fnmatch(url, pattern): # Content is always a dict with format keys (e.g., {"html": "...", "markdown": "..."}) content = content_data.get(format) if content: # Get metadata from cache or use defaults metadata = metadata_cache.get(url, {}) yield CrawlContent( url=url, content=content, status_code=metadata.get('status_code', 200), headers=metadata.get('headers', {}), duration=metadata.get('duration'), log_id=metadata.get('log_id'), country=metadata.get('country'), crawl_uuid=self._uuid ) except Exception: # If contents API fails, yield nothing return def read_batch( self, urls: List[str], formats: List[ContentFormat] = None ) -> Dict[str, Dict[str, str]]: """ Retrieve content for multiple URLs in a single batch request This is more efficient than calling read() multiple times as it retrieves all content in a single API call. Maximum 100 URLs per request. Args: urls: List of URLs to retrieve (max 100) formats: List of content formats to retrieve (e.g., ['markdown', 'text']) If None, defaults to ['html'] Returns: Dictionary mapping URLs to their content in requested formats: { 'https://example.com/page1': { 'markdown': '# Page 1...', 'text': 'Page 1...' }, 'https://example.com/page2': { 'markdown': '# Page 2...', 'text': 'Page 2...' } } Example: ```python # Get markdown and text for multiple URLs urls = ['https://example.com/page1', 'https://example.com/page2'] contents = crawl.read_batch(urls, formats=['markdown', 'text']) for url, formats in contents.items(): markdown = formats.get('markdown', '') text = formats.get('text', '') print(f"{url}: {len(markdown)} chars markdown, {len(text)} chars text") ``` Raises: ValueError: If more than 100 URLs are provided ScrapflyCrawlerError: If crawler not started or request fails """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) if len(urls) > 100: raise ValueError("Maximum 100 URLs per batch request") if not urls: return {} # Default to html if no formats specified if formats is None: formats = ['html'] # Build URL with formats parameter formats_str = ','.join(formats) url = f"{self._client.host}/crawl/{self._uuid}/contents/batch" params = { 'key': self._client.key, 'formats': formats_str } # Prepare request body (newline-separated URLs) body = '\n'.join(urls) # Make request import requests response = requests.post( url, params=params, data=body.encode('utf-8'), headers={'Content-Type': 'text/plain'}, verify=self._client.verify ) if response.status_code != 200: raise ScrapflyCrawlerError( message=f"Batch content request failed: {response.status_code}", code="BATCH_REQUEST_FAILED", http_status_code=response.status_code ) # Parse multipart response content_type = response.headers.get('Content-Type', '') if not content_type.startswith('multipart/related'): raise ScrapflyCrawlerError( message=f"Unexpected content type: {content_type}", code="INVALID_RESPONSE", http_status_code=500 ) # Extract boundary from Content-Type header boundary = None for part in content_type.split(';'): part = part.strip() if part.startswith('boundary='): boundary = part.split('=', 1)[1] break if not boundary: raise ScrapflyCrawlerError( message="No boundary found in multipart response", code="INVALID_RESPONSE", http_status_code=500 ) # Parse multipart message # Prepend Content-Type header to make it a valid email message for the parser message_bytes = f"Content-Type: {content_type}\r\n\r\n".encode('utf-8') + response.content parser = BytesParser(policy=default) message = parser.parsebytes(message_bytes) # Extract content from each part result = {} for part in message.walk(): # Skip the container itself if part.get_content_maintype() == 'multipart': continue # Get the URL from Content-Location header content_location = part.get('Content-Location') if not content_location: continue # Get content type to determine format part_content_type = part.get_content_type() format_type = None # Map MIME types to format names if 'markdown' in part_content_type: format_type = 'markdown' elif 'plain' in part_content_type: format_type = 'text' elif 'html' in part_content_type: format_type = 'html' elif 'json' in part_content_type: format_type = 'json' if not format_type: continue # Get content content = part.get_content() if isinstance(content, bytes): content = content.decode('utf-8', errors='replace') # Initialize URL dict if needed if content_location not in result: result[content_location] = {} # Store content result[content_location][format_type] = content return result def stats(self) -> Dict[str, Any]: """ Get comprehensive statistics about the crawl Returns: Dictionary with crawl statistics Example: ```python stats = crawl.stats() print(f"URLs extracted: {stats['urls_extracted']}") print(f"URLs visited: {stats['urls_visited']}") print(f"Crawl rate: {stats['crawl_rate']:.1f}%") print(f"Total size: {stats['total_size_kb']:.2f} KB") ``` """ status = self.status(refresh=False) # Basic stats from status — uses the wire field names as defined by # Scrapfly source of truth. stats_dict = { 'uuid': self._uuid, 'status': status.status, 'urls_extracted': status.state.urls_extracted, 'urls_visited': status.state.urls_visited, 'urls_to_crawl': status.state.urls_to_crawl, 'urls_failed': status.state.urls_failed, 'urls_skipped': status.state.urls_skipped, 'progress_pct': status.progress_pct, 'is_complete': status.is_complete, 'is_running': status.is_running, 'is_failed': status.is_failed, } # Calculate basic crawl rate (visited vs extracted) if status.state.urls_extracted > 0: stats_dict['crawl_rate'] = (status.state.urls_visited / status.state.urls_extracted) * 100 # Add artifact stats if available if self._artifact_cache is not None: pages = self._artifact_cache.get_pages() total_size = sum(len(p['content']) for p in pages) avg_size = total_size / len(pages) if pages else 0 stats_dict.update({ 'pages_downloaded': len(pages), 'total_size_bytes': total_size, 'total_size_kb': total_size / 1024, 'total_size_mb': total_size / (1024 * 1024), 'avg_page_size_bytes': avg_size, 'avg_page_size_kb': avg_size / 1024, }) # Calculate download rate (pages vs extracted) if status.state.urls_extracted > 0: stats_dict['download_rate'] = (len(pages) / status.state.urls_extracted) * 100 return stats_dict def __repr__(self): url = self._config._params['url'] if self._uuid is None: return f"Crawl(not started, url={url})" status_str = "unknown" if self._status_cache: status_str = self._status_cache.status return f"Crawl(uuid={self._uuid}, url={url}, status={status_str})"High-level abstraction for managing a crawler job
The Crawl object maintains the state of a crawler job and provides convenient methods for managing its lifecycle.
Example
from scrapfly import ScrapflyClient, CrawlerConfig, Crawl client = ScrapflyClient(key='your-key') config = CrawlerConfig(url='https://example.com', page_limit=10) # Create and start crawl crawl = Crawl(client, config) crawl.crawl() # Start the crawler # Wait for completion crawl.wait() # Get results pages = crawl.warc().get_pages() for page in pages: print(f"{page['url']}: {page['status_code']}") # Or read specific URLs html = crawl.read('https://example.com/page1', format='html')Initialize a Crawl object
Args
client- ScrapflyClient instance
config- CrawlerConfig with crawler settings
Instance variables
prop started : bool-
Expand source code
@property def started(self) -> bool: """Check if the crawler has been started""" return self._uuid is not NoneCheck if the crawler has been started
prop uuid : str | None-
Expand source code
@property def uuid(self) -> Optional[str]: """Get the crawler job UUID (None if not started)""" return self._uuidGet the crawler job UUID (None if not started)
Methods
def cancel(self) ‑> bool-
Expand source code
def cancel(self) -> bool: """ Cancel the running crawler job Returns: True if cancelled successfully Raises: ScrapflyCrawlerError: If crawler not started yet Example: ```python # Start a crawl crawl = Crawl(client, config).crawl() # Cancel it crawl.cancel() ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) return self._client.cancel_crawl(self._uuid)Cancel the running crawler job
Returns
True if cancelled successfully
Raises
ScrapflyCrawlerError- If crawler not started yet
Example
# Start a crawl crawl = Crawl(client, config).crawl() # Cancel it crawl.cancel() def crawl(self) ‑> Crawl-
Expand source code
def crawl(self) -> 'Crawl': """ Start the crawler job Returns: Self for method chaining Raises: RuntimeError: If crawler already started Example: ```python crawl = Crawl(client, config) crawl.crawl() # Start crawling ``` """ if self._uuid is not None: raise ScrapflyCrawlerError( message="Crawler already started", code="ALREADY_STARTED", http_status_code=400 ) response = self._client.start_crawl(self._config) self._uuid = response.uuid return selfStart the crawler job
Returns
Self for method chaining
Raises
RuntimeError- If crawler already started
Example
crawl = Crawl(client, config) crawl.crawl() # Start crawling def har(self) ‑> CrawlerArtifactResponse-
Expand source code
def har(self) -> CrawlerArtifactResponse: """ Download the crawler artifact in HAR (HTTP Archive) format Returns: CrawlerArtifactResponse with parsed HAR data Raises: RuntimeError: If crawler not started yet Example: ```python # Get HAR artifact artifact = crawl.har() # Get all pages pages = artifact.get_pages() # Iterate through HAR entries for entry in artifact.iter_responses(): print(f"{entry.url}: {entry.status_code}") print(f"Timing: {entry.time}ms") ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) return self._client.get_crawl_artifact( self._uuid, artifact_type='har' )Download the crawler artifact in HAR (HTTP Archive) format
Returns
CrawlerArtifactResponse with parsed HAR data
Raises
RuntimeError- If crawler not started yet
Example
# Get HAR artifact artifact = crawl.har() # Get all pages pages = artifact.get_pages() # Iterate through HAR entries for entry in artifact.iter_responses(): print(f"{entry.url}: {entry.status_code}") print(f"Timing: {entry.time}ms") def prompt(self,
prompt: str,
search: Dict[str, Any] | None = None,
model: str | None = None,
stream: bool = True) ‑> Iterator[CrawlerPromptEvent] | Dict[str, Any]-
Expand source code
def prompt( self, prompt: str, search: Optional[Dict[str, Any]] = None, model: Optional[str] = None, stream: bool = True, ) -> Union[Iterator[CrawlerPromptEvent], Dict[str, Any]]: """ Ask a question answered from this crawl's content. Sugar over :meth:`ScrapflyClient.crawl_prompt` with a one-element crawl list. Args: prompt: The question. search: Optional retrieval overrides: 'limit', 'mode', 'filters'. model: Optional Gemini model id; unset uses the server default. stream: Yield SSE frames (True) or return one dict (False). Returns: Iterator[CrawlerPromptEvent] when streaming, else Dict. Raises: ScrapflyCrawlerError: if the crawler has not been started yet. CrawlerPromptError: on a server-sent error frame, which can arrive after tokens have already been yielded. Example: ```python for event in crawl.prompt('What does this site sell?'): if event.is_token: print(event.data, end='', flush=True) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400, ) return self._client.crawl_prompt( crawl_ids=[self._uuid], prompt=prompt, search=search, model=model, stream=stream, )Ask a question answered from this crawl's content.
Sugar over :meth:
ScrapflyClient.crawl_prompt()with a one-element crawl list.Args
prompt- The question.
search- Optional retrieval overrides: 'limit', 'mode', 'filters'.
model- Optional Gemini model id; unset uses the server default.
stream- Yield SSE frames (True) or return one dict (False).
Returns
Iterator[CrawlerPromptEvent] when streaming, else Dict.
Raises
ScrapflyCrawlerError- if the crawler has not been started yet.
CrawlerPromptError- on a server-sent error frame, which can arrive after tokens have already been yielded.
Example
for event in crawl.prompt('What does this site sell?'): if event.is_token: print(event.data, end='', flush=True) def read(self,
url: str,
format: Literal['html', 'clean_html', 'markdown', 'json', 'text', 'extracted_data', 'page_metadata'] = 'html') ‑> CrawlContent | None-
Expand source code
def read(self, url: str, format: ContentFormat = 'html') -> Optional[CrawlContent]: """ Read content from a specific URL in the crawl results Args: url: The URL to retrieve content for format: Content format - 'html', 'markdown', 'text', 'clean_html', 'json', 'extracted_data', 'page_metadata' Returns: CrawlContent object with content and metadata, or None if URL not found Example: ```python # Get HTML content for a specific URL content = crawl.read('https://example.com/page1') if content: print(f"URL: {content.url}") print(f"Status: {content.status_code}") print(f"Duration: {content.duration}s") print(content.content) # Get markdown content content = crawl.read('https://example.com/page1', format='markdown') if content: print(content.content) # Check if URL was crawled if crawl.read('https://example.com/missing') is None: print("URL not found in crawl results") ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) # For HTML format, we can get it from the WARC artifact (faster) if format == 'html': artifact = self.warc() for record in artifact.iter_responses(): if record.url == url: # Extract metadata from WARC headers warc_headers = record.warc_headers or {} duration_str = warc_headers.get('WARC-Scrape-Duration') duration = float(duration_str) if duration_str else None return CrawlContent( url=record.url, content=record.content.decode('utf-8', errors='replace'), status_code=record.status_code, headers=record.headers, duration=duration, log_id=warc_headers.get('WARC-Scrape-Log-Id'), country=warc_headers.get('WARC-Scrape-Country'), crawl_uuid=self._uuid ) return None # For other formats (markdown, text, etc.), use the contents API try: result = self._client.get_crawl_contents( self._uuid, format=format ) # The API returns: {"contents": {url: {format: content, ...}, ...}, "links": {...}} contents = result.get('contents', {}) if url in contents: content_data = contents[url] # Content is always a dict with format keys (e.g., {"html": "...", "markdown": "..."}) content_str = content_data.get(format) if content_str: # For non-HTML formats from contents API, we don't have full metadata # Try to get status code from WARC if possible status_code = 200 # Default headers = {} duration = None log_id = None country = None # Try to get metadata from WARC try: artifact = self.warc() for record in artifact.iter_responses(): if record.url == url: status_code = record.status_code headers = record.headers warc_headers = record.warc_headers or {} duration_str = warc_headers.get('WARC-Scrape-Duration') duration = float(duration_str) if duration_str else None log_id = warc_headers.get('WARC-Scrape-Log-Id') country = warc_headers.get('WARC-Scrape-Country') break except: pass return CrawlContent( url=url, content=content_str, status_code=status_code, headers=headers, duration=duration, log_id=log_id, country=country, crawl_uuid=self._uuid ) return None except Exception: # If contents API fails, return None return NoneRead content from a specific URL in the crawl results
Args
url- The URL to retrieve content for
format- Content format - 'html', 'markdown', 'text', 'clean_html', 'json', 'extracted_data', 'page_metadata'
Returns
CrawlContent object with content and metadata, or None if URL not found
Example
# Get HTML content for a specific URL content = crawl.read('https://example.com/page1') if content: print(f"URL: {content.url}") print(f"Status: {content.status_code}") print(f"Duration: {content.duration}s") print(content.content) # Get markdown content content = crawl.read('https://example.com/page1', format='markdown') if content: print(content.content) # Check if URL was crawled if crawl.read('https://example.com/missing') is None: print("URL not found in crawl results") def read_batch(self,
urls: List[str],
formats: List[Literal['html', 'clean_html', 'markdown', 'json', 'text', 'extracted_data', 'page_metadata']] = None) ‑> Dict[str, Dict[str, str]]-
Expand source code
def read_batch( self, urls: List[str], formats: List[ContentFormat] = None ) -> Dict[str, Dict[str, str]]: """ Retrieve content for multiple URLs in a single batch request This is more efficient than calling read() multiple times as it retrieves all content in a single API call. Maximum 100 URLs per request. Args: urls: List of URLs to retrieve (max 100) formats: List of content formats to retrieve (e.g., ['markdown', 'text']) If None, defaults to ['html'] Returns: Dictionary mapping URLs to their content in requested formats: { 'https://example.com/page1': { 'markdown': '# Page 1...', 'text': 'Page 1...' }, 'https://example.com/page2': { 'markdown': '# Page 2...', 'text': 'Page 2...' } } Example: ```python # Get markdown and text for multiple URLs urls = ['https://example.com/page1', 'https://example.com/page2'] contents = crawl.read_batch(urls, formats=['markdown', 'text']) for url, formats in contents.items(): markdown = formats.get('markdown', '') text = formats.get('text', '') print(f"{url}: {len(markdown)} chars markdown, {len(text)} chars text") ``` Raises: ValueError: If more than 100 URLs are provided ScrapflyCrawlerError: If crawler not started or request fails """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) if len(urls) > 100: raise ValueError("Maximum 100 URLs per batch request") if not urls: return {} # Default to html if no formats specified if formats is None: formats = ['html'] # Build URL with formats parameter formats_str = ','.join(formats) url = f"{self._client.host}/crawl/{self._uuid}/contents/batch" params = { 'key': self._client.key, 'formats': formats_str } # Prepare request body (newline-separated URLs) body = '\n'.join(urls) # Make request import requests response = requests.post( url, params=params, data=body.encode('utf-8'), headers={'Content-Type': 'text/plain'}, verify=self._client.verify ) if response.status_code != 200: raise ScrapflyCrawlerError( message=f"Batch content request failed: {response.status_code}", code="BATCH_REQUEST_FAILED", http_status_code=response.status_code ) # Parse multipart response content_type = response.headers.get('Content-Type', '') if not content_type.startswith('multipart/related'): raise ScrapflyCrawlerError( message=f"Unexpected content type: {content_type}", code="INVALID_RESPONSE", http_status_code=500 ) # Extract boundary from Content-Type header boundary = None for part in content_type.split(';'): part = part.strip() if part.startswith('boundary='): boundary = part.split('=', 1)[1] break if not boundary: raise ScrapflyCrawlerError( message="No boundary found in multipart response", code="INVALID_RESPONSE", http_status_code=500 ) # Parse multipart message # Prepend Content-Type header to make it a valid email message for the parser message_bytes = f"Content-Type: {content_type}\r\n\r\n".encode('utf-8') + response.content parser = BytesParser(policy=default) message = parser.parsebytes(message_bytes) # Extract content from each part result = {} for part in message.walk(): # Skip the container itself if part.get_content_maintype() == 'multipart': continue # Get the URL from Content-Location header content_location = part.get('Content-Location') if not content_location: continue # Get content type to determine format part_content_type = part.get_content_type() format_type = None # Map MIME types to format names if 'markdown' in part_content_type: format_type = 'markdown' elif 'plain' in part_content_type: format_type = 'text' elif 'html' in part_content_type: format_type = 'html' elif 'json' in part_content_type: format_type = 'json' if not format_type: continue # Get content content = part.get_content() if isinstance(content, bytes): content = content.decode('utf-8', errors='replace') # Initialize URL dict if needed if content_location not in result: result[content_location] = {} # Store content result[content_location][format_type] = content return resultRetrieve content for multiple URLs in a single batch request
This is more efficient than calling read() multiple times as it retrieves all content in a single API call. Maximum 100 URLs per request.
Args
urls- List of URLs to retrieve (max 100)
formats- List of content formats to retrieve (e.g., ['markdown', 'text']) If None, defaults to ['html']
Returns
Dictionary mapping URLs to their content in requested formats: { 'https://example.com/page1': { 'markdown': '# Page 1…', 'text': 'Page 1…' }, 'https://example.com/page2': { 'markdown': '# Page 2…', 'text': 'Page 2…' } }
Example
# Get markdown and text for multiple URLs urls = ['https://example.com/page1', 'https://example.com/page2'] contents = crawl.read_batch(urls, formats=['markdown', 'text']) for url, formats in contents.items(): markdown = formats.get('markdown', '') text = formats.get('text', '') print(f"{url}: {len(markdown)} chars markdown, {len(text)} chars text")Raises
ValueError- If more than 100 URLs are provided
ScrapflyCrawlerError- If crawler not started or request fails
def read_iter(self,
pattern: str,
format: Literal['html', 'clean_html', 'markdown', 'json', 'text', 'extracted_data', 'page_metadata'] = 'html') ‑> Iterator[CrawlContent]-
Expand source code
def read_iter( self, pattern: str, format: ContentFormat = 'html' ) -> Iterator[CrawlContent]: """ Iterate through URLs matching a pattern and yield their content Supports wildcard patterns using * and ? for flexible URL matching. Args: pattern: URL pattern with wildcards (* matches any characters, ? matches one) Examples: "/products?page=*", "https://example.com/*/detail", "*/product/*" format: Content format to retrieve Yields: CrawlContent objects for each matching URL Example: ```python # Get all product pages in markdown for content in crawl.read_iter(pattern="*/products?page=*", format="markdown"): print(f"{content.url}: {len(content.content)} chars") print(f"Duration: {content.duration}s") # Get all detail pages for content in crawl.read_iter(pattern="*/detail/*"): process(content.content) # Pattern matching examples: # "/products?page=*" matches /products?page=1, /products?page=2, etc. # "*/product/*" matches any URL with /product/ in the path # "https://example.com/page?" matches https://example.com/page1, page2, etc. ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) # For HTML format, use WARC artifact (faster) if format == 'html': artifact = self.warc() for record in artifact.iter_responses(): if fnmatch.fnmatch(record.url, pattern): # Extract metadata from WARC headers warc_headers = record.warc_headers or {} duration_str = warc_headers.get('WARC-Scrape-Duration') duration = float(duration_str) if duration_str else None yield CrawlContent( url=record.url, content=record.content.decode('utf-8', errors='replace'), status_code=record.status_code, headers=record.headers, duration=duration, log_id=warc_headers.get('WARC-Scrape-Log-Id'), country=warc_headers.get('WARC-Scrape-Country'), crawl_uuid=self._uuid ) else: # For other formats, use contents API try: result = self._client.get_crawl_contents( self._uuid, format=format ) contents = result.get('contents', {}) # Build a metadata cache from WARC for non-HTML formats metadata_cache = {} try: artifact = self.warc() for record in artifact.iter_responses(): warc_headers = record.warc_headers or {} duration_str = warc_headers.get('WARC-Scrape-Duration') metadata_cache[record.url] = { 'status_code': record.status_code, 'headers': record.headers, 'duration': float(duration_str) if duration_str else None, 'log_id': warc_headers.get('WARC-Scrape-Log-Id'), 'country': warc_headers.get('WARC-Scrape-Country') } except: pass # Iterate through matching URLs for url, content_data in contents.items(): if fnmatch.fnmatch(url, pattern): # Content is always a dict with format keys (e.g., {"html": "...", "markdown": "..."}) content = content_data.get(format) if content: # Get metadata from cache or use defaults metadata = metadata_cache.get(url, {}) yield CrawlContent( url=url, content=content, status_code=metadata.get('status_code', 200), headers=metadata.get('headers', {}), duration=metadata.get('duration'), log_id=metadata.get('log_id'), country=metadata.get('country'), crawl_uuid=self._uuid ) except Exception: # If contents API fails, yield nothing returnIterate through URLs matching a pattern and yield their content
Supports wildcard patterns using * and ? for flexible URL matching.
Args
pattern- URL pattern with wildcards ( matches any characters, ? matches one) Examples: "/products?page=", "https://example.com//detail", "/product/*"
format- Content format to retrieve
Yields
CrawlContent objects for each matching URL
Example
# Get all product pages in markdown for content in crawl.read_iter(pattern="*/products?page=*", format="markdown"): print(f"{content.url}: {len(content.content)} chars") print(f"Duration: {content.duration}s") # Get all detail pages for content in crawl.read_iter(pattern="*/detail/*"): process(content.content) # Pattern matching examples: # "/products?page=*" matches /products?page=1, /products?page=2, etc. # "*/product/*" matches any URL with /product/ in the path # "https://example.com/page?" matches <https://example.com/page1,> page2, etc. def refresh_history(self, limit: int | None = None) ‑> List[CrawlerRefreshEntry]-
Expand source code
def refresh_history(self, limit: Optional[int] = None) -> List[CrawlerRefreshEntry]: """ Read this crawl's refresh timeline, newest last. Sugar over :meth:`ScrapflyClient.crawl_refresh_history`. Args: limit: Keep only the last N rows. Returns: List[CrawlerRefreshEntry] Raises: ScrapflyCrawlerError: if the crawler has not been started yet. Example: ```python for entry in crawl.refresh_history(limit=5): print(entry.at, entry.added, entry.updated, entry.removed) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400, ) return self._client.crawl_refresh_history(self._uuid, limit=limit)Read this crawl's refresh timeline, newest last.
Sugar over :meth:
ScrapflyClient.crawl_refresh_history().Args
limit- Keep only the last N rows.
Returns
List[CrawlerRefreshEntry]
Raises
ScrapflyCrawlerError- if the crawler has not been started yet.
Example
for entry in crawl.refresh_history(limit=5): print(entry.at, entry.added, entry.updated, entry.removed) def refresh_now(self) ‑> CrawlerRefreshState-
Expand source code
def refresh_now(self) -> CrawlerRefreshState: """ Re-scrape this crawl's URLs in place, right now. Sugar over :meth:`ScrapflyClient.crawl_refresh_now`. The crawl keeps its uuid, its artifacts and its search index; only pages whose content changed are re-indexed and pages that disappeared are dropped, so anything already pointing at this crawl keeps working. Returns: CrawlerRefreshState Raises: ScrapflyCrawlerError: if the crawler has not been started yet. Example: ```python state = crawl.refresh_now() print(state.status) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400, ) return self._client.crawl_refresh_now(self._uuid)Re-scrape this crawl's URLs in place, right now.
Sugar over :meth:
ScrapflyClient.crawl_refresh_now(). The crawl keeps its uuid, its artifacts and its search index; only pages whose content changed are re-indexed and pages that disappeared are dropped, so anything already pointing at this crawl keeps working.Returns
CrawlerRefreshState
Raises
ScrapflyCrawlerError- if the crawler has not been started yet.
Example
state = crawl.refresh_now() print(state.status) def refresh_settings(self, enabled: bool | None = None, interval_seconds: int | None = None) ‑> CrawlerRefreshState-
Expand source code
def refresh_settings( self, enabled: Optional[bool] = None, interval_seconds: Optional[int] = None, ) -> CrawlerRefreshState: """ Change this crawl's refresh schedule. Sugar over :meth:`ScrapflyClient.crawl_refresh_settings`. Only what is passed is changed. Args: enabled: Turn auto-refresh on or off. interval_seconds: Period between runs, 3600 to 7776000. Returns: CrawlerRefreshState Raises: ScrapflyCrawlerError: if the crawler has not been started yet. Example: ```python crawl.refresh_settings(enabled=True, interval_seconds=86400) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400, ) return self._client.crawl_refresh_settings( self._uuid, enabled=enabled, interval_seconds=interval_seconds, )Change this crawl's refresh schedule.
Sugar over :meth:
ScrapflyClient.crawl_refresh_settings(). Only what is passed is changed.Args
enabled- Turn auto-refresh on or off.
interval_seconds- Period between runs, 3600 to 7776000.
Returns
CrawlerRefreshState
Raises
ScrapflyCrawlerError- if the crawler has not been started yet.
Example
crawl.refresh_settings(enabled=True, interval_seconds=86400) def search(self,
query: str,
limit: int = 10,
mode: Literal['vector', 'fts', 'hybrid'] = 'hybrid',
filters: Dict[str, Any] | None = None,
cursor: str | None = None) ‑> CrawlerSearchResponse-
Expand source code
def search( self, query: str, limit: int = 10, mode: Literal['vector', 'fts', 'hybrid'] = 'hybrid', filters: Optional[Dict[str, Any]] = None, cursor: Optional[str] = None, ) -> CrawlerSearchResponse: """ Search this crawl's index. Sugar over :meth:`ScrapflyClient.crawl_search` with a one-element crawl list. The cross-crawl call is the real endpoint; this shares its implementation so single and multi-crawl search cannot drift. Requires the crawl to have been started with ``CrawlerConfig(search=True)`` and its index to have reached ``READY``/``PARTIAL``; poll :attr:`CrawlerStatusResponse.search` or subscribe to the ``crawler_search_ready`` webhook. An index that is not ready yet comes back in ``response.skipped``, not as an error. Args: query: Free-text query. limit: Maximum results, 1-50 (server cap). mode: 'vector', 'fts' or 'hybrid'. filters: Optional flat filter map (see ``crawl_search``). cursor: Opaque next-page token from a previous response. Returns: CrawlerSearchResponse Raises: ScrapflyCrawlerError: if the crawler has not been started yet. Example: ```python crawl = Crawl(client, CrawlerConfig(url='https://example.com', search=True)).crawl().wait() for hit in crawl.search('pricing', limit=5): print(hit.rank, hit.url) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400, ) return self._client.crawl_search( crawl_ids=[self._uuid], query=query, limit=limit, mode=mode, filters=filters, cursor=cursor, )Search this crawl's index.
Sugar over :meth:
ScrapflyClient.crawl_search()with a one-element crawl list. The cross-crawl call is the real endpoint; this shares its implementation so single and multi-crawl search cannot drift.Requires the crawl to have been started with
CrawlerConfig(search=True)and its index to have reachedREADY/PARTIAL; poll :attr:CrawlerStatusResponse.searchor subscribe to thecrawler_search_readywebhook. An index that is not ready yet comes back inresponse.skipped, not as an error.Args
query- Free-text query.
limit- Maximum results, 1-50 (server cap).
mode- 'vector', 'fts' or 'hybrid'.
filters- Optional flat filter map (see
crawl_search). cursor- Opaque next-page token from a previous response.
Returns
CrawlerSearchResponse
Raises
ScrapflyCrawlerError- if the crawler has not been started yet.
Example
crawl = Crawl(client, CrawlerConfig(url='https://example.com', search=True)).crawl().wait() for hit in crawl.search('pricing', limit=5): print(hit.rank, hit.url) def stats(self) ‑> Dict[str, Any]-
Expand source code
def stats(self) -> Dict[str, Any]: """ Get comprehensive statistics about the crawl Returns: Dictionary with crawl statistics Example: ```python stats = crawl.stats() print(f"URLs extracted: {stats['urls_extracted']}") print(f"URLs visited: {stats['urls_visited']}") print(f"Crawl rate: {stats['crawl_rate']:.1f}%") print(f"Total size: {stats['total_size_kb']:.2f} KB") ``` """ status = self.status(refresh=False) # Basic stats from status — uses the wire field names as defined by # Scrapfly source of truth. stats_dict = { 'uuid': self._uuid, 'status': status.status, 'urls_extracted': status.state.urls_extracted, 'urls_visited': status.state.urls_visited, 'urls_to_crawl': status.state.urls_to_crawl, 'urls_failed': status.state.urls_failed, 'urls_skipped': status.state.urls_skipped, 'progress_pct': status.progress_pct, 'is_complete': status.is_complete, 'is_running': status.is_running, 'is_failed': status.is_failed, } # Calculate basic crawl rate (visited vs extracted) if status.state.urls_extracted > 0: stats_dict['crawl_rate'] = (status.state.urls_visited / status.state.urls_extracted) * 100 # Add artifact stats if available if self._artifact_cache is not None: pages = self._artifact_cache.get_pages() total_size = sum(len(p['content']) for p in pages) avg_size = total_size / len(pages) if pages else 0 stats_dict.update({ 'pages_downloaded': len(pages), 'total_size_bytes': total_size, 'total_size_kb': total_size / 1024, 'total_size_mb': total_size / (1024 * 1024), 'avg_page_size_bytes': avg_size, 'avg_page_size_kb': avg_size / 1024, }) # Calculate download rate (pages vs extracted) if status.state.urls_extracted > 0: stats_dict['download_rate'] = (len(pages) / status.state.urls_extracted) * 100 return stats_dictGet comprehensive statistics about the crawl
Returns
Dictionary with crawl statistics
Example
stats = crawl.stats() print(f"URLs extracted: {stats['urls_extracted']}") print(f"URLs visited: {stats['urls_visited']}") print(f"Crawl rate: {stats['crawl_rate']:.1f}%") print(f"Total size: {stats['total_size_kb']:.2f} KB") def status(self, refresh: bool = True) ‑> CrawlerStatusResponse-
Expand source code
def status(self, refresh: bool = True) -> CrawlerStatusResponse: """ Get current crawler status Args: refresh: If True, fetch fresh status from API. If False, return cached status. Returns: CrawlerStatusResponse with current status Raises: RuntimeError: If crawler not started yet Example: ```python status = crawl.status() print(f"Progress: {status.progress_pct}%") print(f"URLs visited: {status.state.urls_visited}") ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) if refresh or self._status_cache is None: self._status_cache = self._client.get_crawl_status(self._uuid) return self._status_cacheGet current crawler status
Args
refresh- If True, fetch fresh status from API. If False, return cached status.
Returns
CrawlerStatusResponse with current status
Raises
RuntimeError- If crawler not started yet
Example
status = crawl.status() print(f"Progress: {status.progress_pct}%") print(f"URLs visited: {status.state.urls_visited}") def urls(self,
status: Literal['visited', 'pending', 'failed', 'skipped'] | None = None,
page: int = 1,
per_page: int = 100) ‑> CrawlerUrlsResponse-
Expand source code
def urls( self, status: Optional[Literal['visited', 'pending', 'failed', 'skipped']] = None, page: int = 1, per_page: int = 100, ) -> CrawlerUrlsResponse: """ List the crawled URLs, optionally filtered by status. Convenience wrapper around :meth:`ScrapflyClient.get_crawl_urls` that pre-fills the crawler UUID. Args: status: Filter by URL status: 'visited', 'pending', 'failed' or 'skipped'. When None, the server defaults to 'visited'. page: 1-based page number, echoed on the response. per_page: Page size, echoed on the response. Returns: CrawlerUrlsResponse with the URL records and the echoed pagination. Raises: ScrapflyCrawlerError: if the crawler has not been started yet. Example: ```python crawl = Crawl(client, config).crawl().wait() for entry in crawl.urls(status='visited'): print(f"{entry.url} ({entry.status})") ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400, ) return self._client.get_crawl_urls( uuid=self._uuid, status=status, page=page, per_page=per_page, )List the crawled URLs, optionally filtered by status.
Convenience wrapper around :meth:
ScrapflyClient.get_crawl_urls()that pre-fills the crawler UUID.Args
status- Filter by URL status: 'visited', 'pending', 'failed' or 'skipped'. When None, the server defaults to 'visited'.
page- 1-based page number, echoed on the response.
per_page- Page size, echoed on the response.
Returns
CrawlerUrlsResponse with the URL records and the echoed pagination.
Raises
ScrapflyCrawlerError- if the crawler has not been started yet.
Example
crawl = Crawl(client, config).crawl().wait() for entry in crawl.urls(status='visited'): print(f"{entry.url} ({entry.status})") def wait(self,
poll_interval: int = 5,
max_wait: int | None = None,
verbose: bool = False,
allow_cancelled: bool = False) ‑> Crawl-
Expand source code
def wait( self, poll_interval: int = 5, max_wait: Optional[int] = None, verbose: bool = False, allow_cancelled: bool = False, ) -> 'Crawl': """ Wait for crawler to complete Polls the status endpoint until the crawler finishes. Args: poll_interval: Seconds between status checks (default: 5) max_wait: Maximum seconds to wait (None = wait forever) verbose: If True, print progress updates allow_cancelled: If True, return normally when the crawler reaches CANCELLED instead of raising. Useful for the cancel-then-wait pattern where the caller already knows they triggered the cancellation. Defaults to False (raises ScrapflyCrawlerError with code='CANCELLED' on user_cancelled), preserving prior behavior for callers that observe external cancellations. Returns: Self for method chaining Raises: ScrapflyCrawlerError: If crawler not started, failed, or timed out. Also raised on cancellation when ``allow_cancelled=False``. Example: ```python # Wait with progress updates crawl.crawl().wait(verbose=True) # Wait with timeout crawl.crawl().wait(max_wait=300) # 5 minutes max # Cancel from the same call site, then wait without re-raising crawl.cancel() crawl.wait(allow_cancelled=True) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) start_time = time.time() poll_count = 0 while True: status = self.status(refresh=True) poll_count += 1 if verbose: logger.info(f"Poll #{poll_count}: {status.status} - " f"{status.progress_pct:.1f}% - " f"{status.state.urls_visited}/{status.state.urls_extracted} URLs") if status.is_complete: if verbose: logger.info(f"✓ Crawler completed successfully!") return self elif status.is_failed: raise ScrapflyCrawlerError( message=f"Crawler failed with status: {status.status}", code="FAILED", http_status_code=400 ) elif status.is_cancelled: if allow_cancelled: if verbose: logger.info("Crawler was cancelled (allow_cancelled=True)") return self raise ScrapflyCrawlerError( message="Crawler was cancelled", code="CANCELLED", http_status_code=400 ) # Check timeout if max_wait is not None: elapsed = time.time() - start_time if elapsed > max_wait: raise ScrapflyCrawlerError( message=f"Timeout waiting for crawler (>{max_wait}s)", code="TIMEOUT", http_status_code=400 ) time.sleep(poll_interval)Wait for crawler to complete
Polls the status endpoint until the crawler finishes.
Args
poll_interval- Seconds between status checks (default: 5)
max_wait- Maximum seconds to wait (None = wait forever)
verbose- If True, print progress updates
allow_cancelled- If True, return normally when the crawler reaches CANCELLED instead of raising. Useful for the cancel-then-wait pattern where the caller already knows they triggered the cancellation. Defaults to False (raises ScrapflyCrawlerError with code='CANCELLED' on user_cancelled), preserving prior behavior for callers that observe external cancellations.
Returns
Self for method chaining
Raises
ScrapflyCrawlerError- If crawler not started, failed, or timed out.
Also raised on cancellation when
allow_cancelled=False.
Example
# Wait with progress updates crawl.crawl().wait(verbose=True) # Wait with timeout crawl.crawl().wait(max_wait=300) # 5 minutes max # Cancel from the same call site, then wait without re-raising crawl.cancel() crawl.wait(allow_cancelled=True) def warc(self, artifact_type: str = 'warc') ‑> CrawlerArtifactResponse-
Expand source code
def warc(self, artifact_type: str = 'warc') -> CrawlerArtifactResponse: """ Download the crawler artifact (WARC file) Args: artifact_type: Type of artifact to download (default: 'warc') Returns: CrawlerArtifactResponse with parsed WARC data Raises: RuntimeError: If crawler not started yet Example: ```python # Get WARC artifact artifact = crawl.warc() # Get all pages pages = artifact.get_pages() # Iterate through responses for record in artifact.iter_responses(): print(record.url) ``` """ if self._uuid is None: raise ScrapflyCrawlerError( message="Crawler not started yet. Call crawl() first.", code="NOT_STARTED", http_status_code=400 ) if self._artifact_cache is None: self._artifact_cache = self._client.get_crawl_artifact( self._uuid, artifact_type=artifact_type ) return self._artifact_cacheDownload the crawler artifact (WARC file)
Args
artifact_type- Type of artifact to download (default: 'warc')
Returns
CrawlerArtifactResponse with parsed WARC data
Raises
RuntimeError- If crawler not started yet
Example
# Get WARC artifact artifact = crawl.warc() # Get all pages pages = artifact.get_pages() # Iterate through responses for record in artifact.iter_responses(): print(record.url)
class CrawlContent (url: str,
content: str,
status_code: int,
headers: Dict[str, str] | None = None,
duration: float | None = None,
log_id: str | None = None,
country: str | None = None,
crawl_uuid: str | None = None)-
Expand source code
class CrawlContent: """ Response object for a single crawled URL Provides access to content and metadata for a crawled page. Similar to ScrapeApiResponse but for crawler results. Attributes: url: The crawled URL (mandatory) content: Page content in requested format (mandatory) status_code: HTTP response status code (mandatory) headers: HTTP response headers (optional) duration: Request duration in seconds (optional) log_id: Scrape log ID for debugging (optional) log_url: URL to view scrape logs (optional) country: Country the request was made from (optional) Example: ```python # Get content for a URL content = crawl.read('https://example.com', format='markdown') print(f"URL: {content.url}") print(f"Status: {content.status_code}") print(f"Duration: {content.duration}s") print(f"Content: {content.content}") # Access metadata if content.log_url: print(f"View logs: {content.log_url}") ``` """ def __init__( self, url: str, content: str, status_code: int, headers: Optional[Dict[str, str]] = None, duration: Optional[float] = None, log_id: Optional[str] = None, country: Optional[str] = None, crawl_uuid: Optional[str] = None ): """ Initialize CrawlContent Args: url: The crawled URL content: Page content in requested format status_code: HTTP response status code headers: HTTP response headers duration: Request duration in seconds log_id: Scrape log ID country: Country the request was made from crawl_uuid: Crawl job UUID """ self.url = url self.content = content self.status_code = status_code self.headers = headers or {} self.duration = duration self.log_id = log_id self.country = country self._crawl_uuid = crawl_uuid @property def log_url(self) -> Optional[str]: """ Get URL to view scrape logs Returns: Log URL if log_id is available, None otherwise """ if self.log_id: return f"https://scrapfly.io/dashboard/logs/{self.log_id}" return None @property def success(self) -> bool: """Check if the request was successful (2xx status code)""" return 200 <= self.status_code < 300 @property def error(self) -> bool: """Check if the request resulted in an error (4xx/5xx status code)""" return self.status_code >= 400 def __repr__(self) -> str: return (f"CrawlContent(url={self.url!r}, status={self.status_code}, " f"content_length={len(self.content)})") def __str__(self) -> str: return self.content def __len__(self) -> int: """Get content length""" return len(self.content)Response object for a single crawled URL
Provides access to content and metadata for a crawled page. Similar to ScrapeApiResponse but for crawler results.
Attributes
url- The crawled URL (mandatory)
content- Page content in requested format (mandatory)
status_code- HTTP response status code (mandatory)
headers- HTTP response headers (optional)
duration- Request duration in seconds (optional)
log_id- Scrape log ID for debugging (optional)
log_url- URL to view scrape logs (optional)
country- Country the request was made from (optional)
Example
# Get content for a URL content = crawl.read('https://example.com', format='markdown') print(f"URL: {content.url}") print(f"Status: {content.status_code}") print(f"Duration: {content.duration}s") print(f"Content: {content.content}") # Access metadata if content.log_url: print(f"View logs: {content.log_url}")Initialize CrawlContent
Args
url- The crawled URL
content- Page content in requested format
status_code- HTTP response status code
headers- HTTP response headers
duration- Request duration in seconds
log_id- Scrape log ID
country- Country the request was made from
crawl_uuid- Crawl job UUID
Instance variables
prop error : bool-
Expand source code
@property def error(self) -> bool: """Check if the request resulted in an error (4xx/5xx status code)""" return self.status_code >= 400Check if the request resulted in an error (4xx/5xx status code)
prop log_url : str | None-
Expand source code
@property def log_url(self) -> Optional[str]: """ Get URL to view scrape logs Returns: Log URL if log_id is available, None otherwise """ if self.log_id: return f"https://scrapfly.io/dashboard/logs/{self.log_id}" return NoneGet URL to view scrape logs
Returns
Log URL if log_id is available, None otherwise
prop success : bool-
Expand source code
@property def success(self) -> bool: """Check if the request was successful (2xx status code)""" return 200 <= self.status_code < 300Check if the request was successful (2xx status code)
class CrawlerArtifactResponse (artifact_data: bytes, artifact_type: str = 'warc')-
Expand source code
class CrawlerArtifactResponse: """ Response from downloading crawler artifacts Returned by ScrapflyClient.get_crawl_artifact() method. Provides high-level access to crawl results with automatic WARC/HAR parsing. Users don't need to understand WARC or HAR format to use this class. Example: ```python # Get WARC artifact (default) artifact = client.get_crawl_artifact(uuid) # Get HAR artifact artifact = client.get_crawl_artifact(uuid, artifact_type='har') # Easy mode: get all pages as dicts pages = artifact.get_pages() for page in pages: print(f"{page['url']}: {page['status_code']}") html = page['content'].decode('utf-8') # Memory-efficient: iterate one page at a time for record in artifact.iter_responses(): print(f"{record.url}: {record.status_code}") process(record.content) # Save to file artifact.save('crawl_results.warc.gz') ``` """ def __init__(self, artifact_data: bytes, artifact_type: str = 'warc'): """ Initialize from artifact data Args: artifact_data: Raw artifact file bytes artifact_type: Type of artifact ('warc' or 'har') """ self._artifact_data = artifact_data self._artifact_type = artifact_type self._warc_parser: Optional[WarcParser] = None self._har_parser: Optional[HarArchive] = None @property def artifact_type(self) -> str: """Get artifact type ('warc' or 'har')""" return self._artifact_type @property def artifact_data(self) -> bytes: """Get raw artifact data (for advanced users)""" return self._artifact_data @property def warc_data(self) -> bytes: """Get raw WARC data (deprecated, use artifact_data)""" return self._artifact_data @property def parser(self) -> Union[WarcParser, HarArchive]: """Get artifact parser instance (lazy-loaded)""" if self._artifact_type == 'har': if self._har_parser is None: self._har_parser = HarArchive(self._artifact_data) return self._har_parser else: if self._warc_parser is None: self._warc_parser = parse_warc(self._artifact_data) return self._warc_parser def iter_records(self) -> Iterator[Union[WarcRecord, HarEntry]]: """ Iterate through all records For WARC: iterates through all WARC records For HAR: iterates through all HAR entries Yields: WarcRecord or HarEntry: Each record in the artifact """ if self._artifact_type == 'har': return self.parser.iter_entries() else: return self.parser.iter_records() def iter_responses(self) -> Iterator[Union[WarcRecord, HarEntry]]: """ Iterate through HTTP response records only This is more memory-efficient than get_pages() for large crawls. For WARC: iterates through response records For HAR: iterates through all entries (HAR only contains responses) Yields: WarcRecord or HarEntry: HTTP response records with url, status_code, headers, content """ if self._artifact_type == 'har': return self.parser.iter_entries() else: return self.parser.iter_responses() def get_pages(self) -> List[Dict]: """ Get all crawled pages as simple dictionaries This is the easiest way to access crawl results. Works with both WARC and HAR formats. Returns: List of dicts with keys: url, status_code, headers, content Example: ```python pages = artifact.get_pages() for page in pages: print(f"{page['url']}: {len(page['content'])} bytes") html = page['content'].decode('utf-8') ``` """ if self._artifact_type == 'har': # Convert HAR entries to page dicts pages = [] for entry in self.parser.iter_entries(): pages.append({ 'url': entry.url, 'status_code': entry.status_code, 'headers': entry.response_headers, 'content': entry.content }) return pages else: return self.parser.get_pages() @property def total_pages(self) -> int: """Get total number of pages in the artifact""" return len(self.get_pages()) def save(self, filepath: str): """ Save WARC data to file Args: filepath: Path to save the WARC file Example: ```python artifact.save('crawl_results.warc.gz') ``` """ with open(filepath, 'wb') as f: f.write(self.warc_data) def __repr__(self): return f"CrawlerArtifactResponse(size={len(self.warc_data)} bytes)"Response from downloading crawler artifacts
Returned by ScrapflyClient.get_crawl_artifact() method.
Provides high-level access to crawl results with automatic WARC/HAR parsing. Users don't need to understand WARC or HAR format to use this class.
Example
# Get WARC artifact (default) artifact = client.get_crawl_artifact(uuid) # Get HAR artifact artifact = client.get_crawl_artifact(uuid, artifact_type='har') # Easy mode: get all pages as dicts pages = artifact.get_pages() for page in pages: print(f"{page['url']}: {page['status_code']}") html = page['content'].decode('utf-8') # Memory-efficient: iterate one page at a time for record in artifact.iter_responses(): print(f"{record.url}: {record.status_code}") process(record.content) # Save to file artifact.save('crawl_results.warc.gz')Initialize from artifact data
Args
artifact_data- Raw artifact file bytes
artifact_type- Type of artifact ('warc' or 'har')
Instance variables
prop artifact_data : bytes-
Expand source code
@property def artifact_data(self) -> bytes: """Get raw artifact data (for advanced users)""" return self._artifact_dataGet raw artifact data (for advanced users)
prop artifact_type : str-
Expand source code
@property def artifact_type(self) -> str: """Get artifact type ('warc' or 'har')""" return self._artifact_typeGet artifact type ('warc' or 'har')
prop parser : WarcParser | HarArchive-
Expand source code
@property def parser(self) -> Union[WarcParser, HarArchive]: """Get artifact parser instance (lazy-loaded)""" if self._artifact_type == 'har': if self._har_parser is None: self._har_parser = HarArchive(self._artifact_data) return self._har_parser else: if self._warc_parser is None: self._warc_parser = parse_warc(self._artifact_data) return self._warc_parserGet artifact parser instance (lazy-loaded)
prop total_pages : int-
Expand source code
@property def total_pages(self) -> int: """Get total number of pages in the artifact""" return len(self.get_pages())Get total number of pages in the artifact
prop warc_data : bytes-
Expand source code
@property def warc_data(self) -> bytes: """Get raw WARC data (deprecated, use artifact_data)""" return self._artifact_dataGet raw WARC data (deprecated, use artifact_data)
Methods
def get_pages(self) ‑> List[Dict]-
Expand source code
def get_pages(self) -> List[Dict]: """ Get all crawled pages as simple dictionaries This is the easiest way to access crawl results. Works with both WARC and HAR formats. Returns: List of dicts with keys: url, status_code, headers, content Example: ```python pages = artifact.get_pages() for page in pages: print(f"{page['url']}: {len(page['content'])} bytes") html = page['content'].decode('utf-8') ``` """ if self._artifact_type == 'har': # Convert HAR entries to page dicts pages = [] for entry in self.parser.iter_entries(): pages.append({ 'url': entry.url, 'status_code': entry.status_code, 'headers': entry.response_headers, 'content': entry.content }) return pages else: return self.parser.get_pages()Get all crawled pages as simple dictionaries
This is the easiest way to access crawl results. Works with both WARC and HAR formats.
Returns
Listofdicts with keys- url, status_code, headers, content
Example
pages = artifact.get_pages() for page in pages: print(f"{page['url']}: {len(page['content'])} bytes") html = page['content'].decode('utf-8') def iter_records(self) ‑> Iterator[WarcRecord | HarEntry]-
Expand source code
def iter_records(self) -> Iterator[Union[WarcRecord, HarEntry]]: """ Iterate through all records For WARC: iterates through all WARC records For HAR: iterates through all HAR entries Yields: WarcRecord or HarEntry: Each record in the artifact """ if self._artifact_type == 'har': return self.parser.iter_entries() else: return self.parser.iter_records()Iterate through all records
For WARC: iterates through all WARC records For HAR: iterates through all HAR entries
Yields
WarcRecordorHarEntry- Each record in the artifact
def iter_responses(self) ‑> Iterator[WarcRecord | HarEntry]-
Expand source code
def iter_responses(self) -> Iterator[Union[WarcRecord, HarEntry]]: """ Iterate through HTTP response records only This is more memory-efficient than get_pages() for large crawls. For WARC: iterates through response records For HAR: iterates through all entries (HAR only contains responses) Yields: WarcRecord or HarEntry: HTTP response records with url, status_code, headers, content """ if self._artifact_type == 'har': return self.parser.iter_entries() else: return self.parser.iter_responses()Iterate through HTTP response records only
This is more memory-efficient than get_pages() for large crawls.
For WARC: iterates through response records For HAR: iterates through all entries (HAR only contains responses)
Yields
WarcRecordorHarEntry- HTTP response records with url, status_code, headers, content
def save(self, filepath: str)-
Expand source code
def save(self, filepath: str): """ Save WARC data to file Args: filepath: Path to save the WARC file Example: ```python artifact.save('crawl_results.warc.gz') ``` """ with open(filepath, 'wb') as f: f.write(self.warc_data)Save WARC data to file
Args
filepath- Path to save the WARC file
Example
artifact.save('crawl_results.warc.gz')
class CrawlerConfig (url: str | None = None,
url_list: List[str] | None = None,
remote_url_list: str | None = None,
page_limit: int | None = None,
max_depth: int | None = None,
max_duration: int | None = None,
exclude_paths: List[str] | None = None,
include_only_paths: List[str] | None = None,
ignore_base_path_restriction: bool = False,
follow_external_links: bool = False,
allowed_external_domains: List[str] | None = None,
follow_internal_subdomains: bool | None = None,
allowed_internal_subdomains: List[str] | None = None,
headers: Dict[str, str] | None = None,
delay: int | None = None,
user_agent: str | None = None,
max_concurrency: int | None = None,
rendering_delay: int | None = None,
use_sitemaps: bool = False,
respect_robots_txt: bool | None = None,
ignore_no_follow: bool = False,
cache: bool = False,
cache_ttl: int | None = None,
cache_clear: bool = False,
content_formats: List[Literal['html', 'markdown', 'text', 'clean_html']] | None = None,
extraction_rules: Dict | None = None,
asp: bool | scrapfly.scrape_config._Unset = <unset>,
proxy_pool: str | None = None,
country: str | None = None,
webhook_name: str | None = None,
webhook_events: List[str] | None = None,
max_api_credit: int | None = None,
search: bool = False,
refresh: bool = False,
refresh_interval: int | None = None,
unblocker: bool | scrapfly.scrape_config._Unset = <unset>)-
Expand source code
class CrawlerConfig(BaseApiConfig): """ Configuration for Scrapfly Crawler API The Crawler API performs recursive website crawling with advanced configuration, content extraction, and artifact storage. Example: ```python from scrapfly import ScrapflyClient, CrawlerConfig client = ScrapflyClient(key='YOUR_API_KEY') config = CrawlerConfig( url='https://example.com', page_limit=100, max_depth=3, content_formats=['markdown', 'html'] ) # Start crawl start_response = client.start_crawl(config) uuid = start_response.uuid # Poll status status = client.get_crawl_status(uuid) # Get results when complete if status.is_complete: artifact = client.get_crawl_artifact(uuid) pages = artifact.get_pages() ``` """ WEBHOOK_CRAWLER_STARTED = 'crawler_started' WEBHOOK_CRAWLER_URL_VISITED = 'crawler_url_visited' WEBHOOK_CRAWLER_URL_SKIPPED = 'crawler_url_skipped' WEBHOOK_CRAWLER_URL_DISCOVERED = 'crawler_url_discovered' WEBHOOK_CRAWLER_URL_FAILED = 'crawler_url_failed' WEBHOOK_CRAWLER_STOPPED = 'crawler_stopped' WEBHOOK_CRAWLER_CANCELLED = 'crawler_cancelled' WEBHOOK_CRAWLER_FINISHED = 'crawler_finished' WEBHOOK_CRAWLER_SEARCH_READY = 'crawler_search_ready' WEBHOOK_CRAWLER_SEARCH_FAILED = 'crawler_search_failed' WEBHOOK_CRAWLER_UPDATED = 'crawler_updated' # Auto-refresh interval bounds. The floor decides the cost: a crawl # refreshing every minute re-scrapes the whole site 1,440 times a day. REFRESH_MIN_INTERVAL = 3600 REFRESH_MAX_INTERVAL = 90 * 24 * 3600 ALL_WEBHOOK_EVENTS = [ WEBHOOK_CRAWLER_STARTED, WEBHOOK_CRAWLER_URL_VISITED, WEBHOOK_CRAWLER_URL_SKIPPED, WEBHOOK_CRAWLER_URL_DISCOVERED, WEBHOOK_CRAWLER_URL_FAILED, WEBHOOK_CRAWLER_STOPPED, WEBHOOK_CRAWLER_CANCELLED, WEBHOOK_CRAWLER_FINISHED, WEBHOOK_CRAWLER_SEARCH_READY, WEBHOOK_CRAWLER_SEARCH_FAILED, WEBHOOK_CRAWLER_UPDATED, ] def __init__( self, url: Optional[str] = None, # URL source — exactly one of url, url_list, remote_url_list must be set. # url enables discovery (sitemaps/robots/links); url_list and # remote_url_list crawl an explicit set of URLs with discovery off. url_list: Optional[List[str]] = None, remote_url_list: Optional[str] = None, # Crawl limits page_limit: Optional[int] = None, max_depth: Optional[int] = None, max_duration: Optional[int] = None, # Path filtering (mutually exclusive) exclude_paths: Optional[List[str]] = None, include_only_paths: Optional[List[str]] = None, # Advanced crawl options ignore_base_path_restriction: bool = False, follow_external_links: bool = False, allowed_external_domains: Optional[List[str]] = None, # Subdomain control (NEW — added in 0.8.28 to match the documented public API). # Server-side default for follow_internal_subdomains is True; we leave the # field unset by default so the server applies its own default. follow_internal_subdomains: Optional[bool] = None, allowed_internal_subdomains: Optional[List[str]] = None, # Request configuration headers: Optional[Dict[str, str]] = None, delay: Optional[int] = None, user_agent: Optional[str] = None, max_concurrency: Optional[int] = None, rendering_delay: Optional[int] = None, # Crawl strategy options use_sitemaps: bool = False, # respect_robots_txt: server default is True. Leave unset (None) so the # server applies its own default rather than forcing False on every request. respect_robots_txt: Optional[bool] = None, ignore_no_follow: bool = False, # Cache options cache: bool = False, cache_ttl: Optional[int] = None, cache_clear: bool = False, # Content extraction content_formats: Optional[List[Literal['html', 'markdown', 'text', 'clean_html']]] = None, extraction_rules: Optional[Dict] = None, # Web scraping features asp: Union[bool, _Unset] = _UNSET, # deprecated alias of `unblocker`, which is declared last proxy_pool: Optional[str] = None, country: Optional[str] = None, # Webhook integration webhook_name: Optional[str] = None, webhook_events: Optional[List[str]] = None, # Cost control max_api_credit: Optional[int] = None, # New options follow every legacy positional argument. Inserting them # above asp would reinterpret existing bypass/proxy settings as search # and recurring refresh settings. # Search index built during the crawl, queried through # client.crawl_search() / client.crawl_prompt() once READY. search: bool = False, # Auto-refresh: re-scrape this crawl's own URLs in place, on a period. # Same crawler_uuid, same artifacts, only changed pages re-indexed. refresh: bool = False, refresh_interval: Optional[int] = None, unblocker: Union[bool, _Unset] = _UNSET ): """ Initialize a CrawlerConfig Args: url: Starting URL for the crawl (required) page_limit: Maximum number of pages to crawl max_depth: Maximum crawl depth from starting URL max_duration: Maximum crawl duration in seconds exclude_paths: List of path patterns to exclude (mutually exclusive with include_only_paths) include_only_paths: List of path patterns to include only (mutually exclusive with exclude_paths) ignore_base_path_restriction: Allow crawling outside the base path follow_external_links: Follow links to external domains allowed_external_domains: List of external domains allowed when follow_external_links is True headers: Custom HTTP headers for requests delay: Delay between requests in milliseconds user_agent: Custom user agent string max_concurrency: Maximum concurrent requests rendering_delay: Delay for JavaScript rendering in milliseconds use_sitemaps: Use sitemap.xml to discover URLs respect_robots_txt: Respect robots.txt rules ignore_no_follow: Ignore rel="nofollow" attributes cache: Enable caching cache_ttl: Cache time-to-live in seconds cache_clear: Clear cache before crawling content_formats: List of content formats to extract ('html', 'markdown', 'text', 'clean_html') extraction_rules: Custom extraction rules search: Build a semantic search index while the crawl runs refresh: Keep this crawl fresh by re-scraping its own URLs in place refresh_interval: Seconds between refresh runs (3600 to 7776000) unblocker: Enable the anti-bot bypass (Unblocker) asp: Deprecated alias of `unblocker`, permanently supported. When both are supplied, `asp` wins. proxy_pool: Proxy pool to use (e.g., 'public_residential_pool') country: Target country for geo-located content webhook_name: Webhook name for event notifications webhook_events: List of webhook events to trigger max_api_credit: Maximum API credits to spend on this crawl """ if exclude_paths and include_only_paths: raise ValueError("exclude_paths and include_only_paths are mutually exclusive") if refresh_interval is not None and not (self.REFRESH_MIN_INTERVAL <= refresh_interval <= self.REFRESH_MAX_INTERVAL): raise ValueError( f"refresh_interval must be between {self.REFRESH_MIN_INTERVAL} and {self.REFRESH_MAX_INTERVAL} seconds" ) if refresh_interval is not None and not refresh: raise ValueError("refresh_interval requires refresh=True") sources_set = sum(1 for v in (url, url_list, remote_url_list) if v) if sources_set == 0: raise ValueError("Provide one of: url, url_list, remote_url_list") if sources_set > 1: raise ValueError("Only one of url, url_list, remote_url_list can be set") params: Dict = {} if url: params['url'] = url if url_list: params['url_list'] = url_list if remote_url_list: params['remote_url_list'] = remote_url_list # Add optional parameters if page_limit is not None: params['page_limit'] = page_limit if max_depth is not None: params['max_depth'] = max_depth if max_duration is not None: params['max_duration'] = max_duration # Path filtering if exclude_paths: params['exclude_paths'] = exclude_paths if include_only_paths: params['include_only_paths'] = include_only_paths # Advanced options if ignore_base_path_restriction: params['ignore_base_path_restriction'] = True if follow_external_links: params['follow_external_links'] = True if allowed_external_domains: params['allowed_external_domains'] = allowed_external_domains # Subdomain control (NEW). Both fields are tri-state: None means # "unset" (server default applies); explicit True/False / list overrides. if follow_internal_subdomains is not None: params['follow_internal_subdomains'] = follow_internal_subdomains if allowed_internal_subdomains: params['allowed_internal_subdomains'] = allowed_internal_subdomains # Request configuration if headers: params['headers'] = headers if delay is not None: params['delay'] = delay if user_agent: params['user_agent'] = user_agent if max_concurrency is not None: params['max_concurrency'] = max_concurrency if rendering_delay is not None: params['rendering_delay'] = rendering_delay # Crawl strategy if use_sitemaps: params['use_sitemaps'] = True # Tri-state: None = let server default win (default True). Explicit # True/False overrides. if respect_robots_txt is not None: params['respect_robots_txt'] = respect_robots_txt if ignore_no_follow: params['ignore_no_follow'] = True # Cache if cache: params['cache'] = True if cache_ttl is not None: params['cache_ttl'] = cache_ttl if cache_clear: params['cache_clear'] = True # Content extraction if content_formats: params['content_formats'] = content_formats if extraction_rules: params['extraction_rules'] = extraction_rules # Search index if search: params['search'] = True # Auto-refresh. The interval is omitted when unset so the server # default period applies. if refresh: params['refresh'] = True if refresh_interval is not None: params['refresh_interval'] = refresh_interval # Web scraping features. Both input names collapse here, and the key # emitted to POST /crawl stays `asp`: published SDK versions are # immutable and upgraded per installation, so emitting `unblocker` # against an API deployment that has not learned it yet would silently # drop a paid feature (crawl succeeds, is billed, returns blocked pages). if _resolve_unblocker(asp, unblocker): params['asp'] = True if proxy_pool: params['proxy_pool'] = proxy_pool if country: params['country'] = country # Webhooks if webhook_name: params['webhook_name'] = webhook_name if webhook_events: assert all( event in self.ALL_WEBHOOK_EVENTS for event in webhook_events ), f"Invalid webhook events. Valid events are: {self.ALL_WEBHOOK_EVENTS}" params['webhook_events'] = webhook_events # Cost control if max_api_credit is not None: params['max_api_credit'] = max_api_credit self._params = params @property def unblocker(self) -> bool: """Anti-bot bypass toggle, the current name for what used to be `asp`. Backed by the single `asp` entry of the request body, so reading or writing either name sees the same state. Assigning a falsy value drops the key entirely, which is how "disabled" has always been expressed. """ return bool(self._params.get('asp', False)) @unblocker.setter def unblocker(self, value: bool): if value: self._params['asp'] = True else: self._params.pop('asp', None) @property def asp(self) -> bool: """Deprecated alias of `unblocker`, permanently supported.""" return self.unblocker @asp.setter def asp(self, value: bool): self.unblocker = value def to_api_params(self, key: Optional[str] = None) -> Dict: """ Convert config to API parameters :param key: API key (optional, can be added by client) :return: Dictionary of API parameters """ params = self._params.copy() if key: params['key'] = key return params def to_multipart_parts(self) -> Dict: """ Split the configuration into the two parts required by the ``POST /crawl`` multipart endpoint: - ``config``: a JSON object with every field except ``url_list`` - ``urls``: a newline-delimited text payload, one URL per line (only present when an explicit URL list was provided) :return: dict with keys ``config`` (dict) and ``urls`` (Optional[str]) """ body = self._params.copy() urls_blob: Optional[str] = None if 'url_list' in body: urls = body.pop('url_list') if urls: urls_blob = "\n".join(urls) return {'config': body, 'urls': urls_blob}Configuration for Scrapfly Crawler API
The Crawler API performs recursive website crawling with advanced configuration, content extraction, and artifact storage.
Example
from scrapfly import ScrapflyClient, CrawlerConfig client = ScrapflyClient(key='YOUR_API_KEY') config = CrawlerConfig( url='https://example.com', page_limit=100, max_depth=3, content_formats=['markdown', 'html'] ) # Start crawl start_response = client.start_crawl(config) uuid = start_response.uuid # Poll status status = client.get_crawl_status(uuid) # Get results when complete if status.is_complete: artifact = client.get_crawl_artifact(uuid) pages = artifact.get_pages()Initialize a CrawlerConfig
Args
url- Starting URL for the crawl (required)
page_limit- Maximum number of pages to crawl
max_depth- Maximum crawl depth from starting URL
max_duration- Maximum crawl duration in seconds
exclude_paths- List of path patterns to exclude (mutually exclusive with include_only_paths)
include_only_paths- List of path patterns to include only (mutually exclusive with exclude_paths)
ignore_base_path_restriction- Allow crawling outside the base path
follow_external_links- Follow links to external domains
allowed_external_domains- List of external domains allowed when follow_external_links is True
headers- Custom HTTP headers for requests
delay- Delay between requests in milliseconds
user_agent- Custom user agent string
max_concurrency- Maximum concurrent requests
rendering_delay- Delay for JavaScript rendering in milliseconds
use_sitemaps- Use sitemap.xml to discover URLs
respect_robots_txt- Respect robots.txt rules
ignore_no_follow- Ignore rel="nofollow" attributes
cache- Enable caching
cache_ttl- Cache time-to-live in seconds
cache_clear- Clear cache before crawling
content_formats- List of content formats to extract ('html', 'markdown', 'text', 'clean_html')
extraction_rules- Custom extraction rules
search- Build a semantic search index while the crawl runs
refresh- Keep this crawl fresh by re-scraping its own URLs in place
refresh_interval- Seconds between refresh runs (3600 to 7776000)
unblocker- Enable the anti-bot bypass (Unblocker)
asp- Deprecated alias of
unblocker, permanently supported. When both are supplied,aspwins. proxy_pool- Proxy pool to use (e.g., 'public_residential_pool')
country- Target country for geo-located content
webhook_name- Webhook name for event notifications
webhook_events- List of webhook events to trigger
max_api_credit- Maximum API credits to spend on this crawl
Ancestors
Class variables
var ALL_WEBHOOK_EVENTSvar REFRESH_MAX_INTERVALvar REFRESH_MIN_INTERVALvar WEBHOOK_CRAWLER_CANCELLEDvar WEBHOOK_CRAWLER_FINISHEDvar WEBHOOK_CRAWLER_SEARCH_FAILEDvar WEBHOOK_CRAWLER_SEARCH_READYvar WEBHOOK_CRAWLER_STARTEDvar WEBHOOK_CRAWLER_STOPPEDvar WEBHOOK_CRAWLER_UPDATEDvar WEBHOOK_CRAWLER_URL_DISCOVEREDvar WEBHOOK_CRAWLER_URL_FAILEDvar WEBHOOK_CRAWLER_URL_SKIPPEDvar WEBHOOK_CRAWLER_URL_VISITED
Instance variables
prop asp : bool-
Expand source code
@property def asp(self) -> bool: """Deprecated alias of `unblocker`, permanently supported.""" return self.unblockerDeprecated alias of
unblocker, permanently supported. prop unblocker : bool-
Expand source code
@property def unblocker(self) -> bool: """Anti-bot bypass toggle, the current name for what used to be `asp`. Backed by the single `asp` entry of the request body, so reading or writing either name sees the same state. Assigning a falsy value drops the key entirely, which is how "disabled" has always been expressed. """ return bool(self._params.get('asp', False))Anti-bot bypass toggle, the current name for what used to be
asp.Backed by the single
aspentry of the request body, so reading or writing either name sees the same state. Assigning a falsy value drops the key entirely, which is how "disabled" has always been expressed.
Methods
def to_api_params(self, key: str | None = None) ‑> Dict-
Expand source code
def to_api_params(self, key: Optional[str] = None) -> Dict: """ Convert config to API parameters :param key: API key (optional, can be added by client) :return: Dictionary of API parameters """ params = self._params.copy() if key: params['key'] = key return paramsConvert config to API parameters
:param key: API key (optional, can be added by client) :return: Dictionary of API parameters
def to_multipart_parts(self) ‑> Dict-
Expand source code
def to_multipart_parts(self) -> Dict: """ Split the configuration into the two parts required by the ``POST /crawl`` multipart endpoint: - ``config``: a JSON object with every field except ``url_list`` - ``urls``: a newline-delimited text payload, one URL per line (only present when an explicit URL list was provided) :return: dict with keys ``config`` (dict) and ``urls`` (Optional[str]) """ body = self._params.copy() urls_blob: Optional[str] = None if 'url_list' in body: urls = body.pop('url_list') if urls: urls_blob = "\n".join(urls) return {'config': body, 'urls': urls_blob}Split the configuration into the two parts required by the
POST /crawlmultipart endpoint:config: a JSON object with every field excepturl_listurls: a newline-delimited text payload, one URL per line (only present when an explicit URL list was provided)
:return: dict with keys
config(dict) andurls(Optional[str])
class CrawlerError (message: str,
code: str,
http_status_code: int,
resource: str | None = None,
is_retryable: bool = False,
retry_delay: int | None = None,
retry_times: int | None = None,
documentation_url: str | None = None,
api_response: ForwardRef('ApiResponse') | None = None)-
Expand source code
class CrawlerError(ScrapflyError): """Base exception for Crawler API errors""" passBase exception for Crawler API errors
Ancestors
- ScrapflyError
- builtins.Exception
- builtins.BaseException
Subclasses
class CrawlerLifecycleWebhook (event: str,
crawler_uuid: str,
project: str,
env: str,
action: str,
state: CrawlerState,
seed_url: str,
status_link: str)-
Expand source code
@dataclass class CrawlerLifecycleWebhook(CrawlerWebhookBase): """ Payload for the 4 lifecycle events: ``crawler_started``, ``crawler_stopped``, ``crawler_cancelled``, ``crawler_finished``. These events all carry the same fields: the seed URL, the common base (crawler_uuid / project / env / action / state), and a ``links.status`` URL pointing at the crawl status endpoint. Disambiguate by inspecting ``self.event`` (use :class:`CrawlerWebhookEvent`). Attributes: seed_url: The root URL the crawl was started from. status_link: URL to fetch the live crawler status. """ seed_url: str status_link: str @classmethod def from_payload(cls, event: str, payload: Dict[str, Any]) -> 'CrawlerLifecycleWebhook': base = cls._parse_base(event, payload) return cls( **base, seed_url=payload['seed_url'], status_link=payload['links']['status'], )Payload for the 4 lifecycle events:
crawler_started,crawler_stopped,crawler_cancelled,crawler_finished.These events all carry the same fields: the seed URL, the common base (crawler_uuid / project / env / action / state), and a
links.statusURL pointing at the crawl status endpoint. Disambiguate by inspectingself.event(use :class:CrawlerWebhookEvent).Attributes
seed_url- The root URL the crawl was started from.
status_link- URL to fetch the live crawler status.
Ancestors
Static methods
def from_payload(event: str, payload: Dict[str, Any]) ‑> CrawlerLifecycleWebhook
Instance variables
var seed_url : strvar status_link : str
class CrawlerPromptError (message: str,
code: str,
http_status_code: int,
resource: str | None = None,
is_retryable: bool = False,
retry_delay: int | None = None,
retry_times: int | None = None,
documentation_url: str | None = None,
api_response: ForwardRef('ApiResponse') | None = None)-
Expand source code
class CrawlerPromptError(CrawlerError): """ Exception raised when ``POST /crawl/prompt`` fails. Also raised mid-stream when the server sends an ``event: error`` frame: generation can fail after tokens have already been delivered, so a caller consuming the iterator must be ready for this on any ``next()``. """ passException raised when
POST /crawl/promptfails.Also raised mid-stream when the server sends an
event: errorframe: generation can fail after tokens have already been delivered, so a caller consuming the iterator must be ready for this on anynext().Ancestors
- CrawlerError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class CrawlerPromptEvent (event: str, data: Any)-
Expand source code
@dataclass class CrawlerPromptEvent: """ One frame of the ``POST /crawl/prompt`` SSE stream. Frame order is ``source``* → ``token``* → (``error``) → ``done``. Keepalive comment frames are consumed by the reader and never surfaced. Attributes: event: ``source``, ``token``, ``error`` or ``done``. data: The decoded frame payload: a str for ``token``, a dict for the other three. """ event: str data: Any @property def is_token(self) -> bool: return self.event == 'token' @property def is_done(self) -> bool: return self.event == 'done'One frame of the
POST /crawl/promptSSE stream.Frame order is
source→token→ (error) →done. Keepalive comment frames are consumed by the reader and never surfaced.Attributes
eventsource,token,errorordone.data- The decoded frame payload: a str for
token, a dict for the other three.
Instance variables
var data : Anyvar event : strprop is_done : bool-
Expand source code
@property def is_done(self) -> bool: return self.event == 'done' prop is_token : bool-
Expand source code
@property def is_token(self) -> bool: return self.event == 'token'
class CrawlerRefreshEntry (at: str | None = None,
generation: int | None = None,
added: int = 0,
updated: int = 0,
removed: int = 0,
unchanged: int = 0,
failed: int = 0,
duration_ms: int | None = None,
search_status: str | None = None,
error: str | None = None,
sample_updated: List[str] = <factory>,
sample_removed: List[str] = <factory>)-
Expand source code
@dataclass class CrawlerRefreshEntry: """ One row of a crawl's refresh timeline. Counts describe the whole run; ``sample_updated`` / ``sample_removed`` carry at most ten URLs each. The full URL lists are never inlined, so a 5,000-page crawl does not put 5,000 strings into every status poll. Attributes: at: ISO-8601 timestamp of the run. generation: Refresh generation this run produced, 1 for the first. added: URLs discovered by this run that the crawl did not hold. updated: Known URLs whose content fingerprint changed. removed: Known URLs that no longer exist and were dropped. unchanged: Known URLs re-scraped with an identical fingerprint. These cost no embedding and no index write. failed: URLs the run could not fetch. They keep their previous content. duration_ms: Wall time of the run. search_status: Index status after the run, ``None`` when the crawl has no search index. error: Failure reason when the run itself failed. sample_updated: Up to ten re-indexed URLs. sample_removed: Up to ten dropped URLs. """ at: Optional[str] = None generation: Optional[int] = None added: int = 0 updated: int = 0 removed: int = 0 unchanged: int = 0 failed: int = 0 duration_ms: Optional[int] = None search_status: Optional[str] = None error: Optional[str] = None sample_updated: List[str] = field(default_factory=list) sample_removed: List[str] = field(default_factory=list) @classmethod def from_dict(cls, data: Dict[str, Any]) -> 'CrawlerRefreshEntry': return cls( at=data.get('at'), generation=data.get('generation'), added=data.get('added') or 0, updated=data.get('updated') or 0, removed=data.get('removed') or 0, unchanged=data.get('unchanged') or 0, failed=data.get('failed') or 0, duration_ms=data.get('duration_ms'), search_status=data.get('search_status'), error=data.get('error'), sample_updated=list(data.get('sample_updated') or []), sample_removed=list(data.get('sample_removed') or []), ) @property def changed(self) -> int: """Pages the run actually touched. Zero means the site stood still.""" return self.added + self.updated + self.removedOne row of a crawl's refresh timeline.
Counts describe the whole run;
sample_updated/sample_removedcarry at most ten URLs each. The full URL lists are never inlined, so a 5,000-page crawl does not put 5,000 strings into every status poll.Attributes
at- ISO-8601 timestamp of the run.
generation- Refresh generation this run produced, 1 for the first.
added- URLs discovered by this run that the crawl did not hold.
updated- Known URLs whose content fingerprint changed.
removed- Known URLs that no longer exist and were dropped.
unchanged- Known URLs re-scraped with an identical fingerprint. These cost no embedding and no index write.
failed- URLs the run could not fetch. They keep their previous content.
duration_ms- Wall time of the run.
search_status- Index status after the run,
Nonewhen the crawl has no search index. error- Failure reason when the run itself failed.
sample_updated- Up to ten re-indexed URLs.
sample_removed- Up to ten dropped URLs.
Static methods
def from_dict(data: Dict[str, Any]) ‑> CrawlerRefreshEntry
Instance variables
var added : intvar at : str | Noneprop changed : int-
Expand source code
@property def changed(self) -> int: """Pages the run actually touched. Zero means the site stood still.""" return self.added + self.updated + self.removedPages the run actually touched. Zero means the site stood still.
var duration_ms : int | Nonevar error : str | Nonevar failed : intvar generation : int | Nonevar removed : intvar sample_removed : List[str]var sample_updated : List[str]var search_status : str | Nonevar unchanged : intvar updated : int
class CrawlerRefreshError (message: str,
code: str,
http_status_code: int,
resource: str | None = None,
is_retryable: bool = False,
retry_delay: int | None = None,
retry_times: int | None = None,
documentation_url: str | None = None,
api_response: ForwardRef('ApiResponse') | None = None)-
Expand source code
class CrawlerRefreshError(CrawlerError): """ Exception raised when a crawl refresh call fails. Carries the API code, e.g. ``ERR::CRAWLER::REFRESH_NOT_ENABLED``, ``ERR::CRAWLER::REFRESH_IN_PROGRESS``, ``ERR::CRAWLER::REFRESH_INTERVAL_INVALID``. """ passException raised when a crawl refresh call fails.
Carries the API code, e.g.
ERR::CRAWLER::REFRESH_NOT_ENABLED,ERR::CRAWLER::REFRESH_IN_PROGRESS,ERR::CRAWLER::REFRESH_INTERVAL_INVALID.Ancestors
- CrawlerError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class CrawlerRefreshState (response_data: Dict[str, Any])-
Expand source code
class CrawlerRefreshState: """ The ``refresh`` block of a crawl, as carried by ``GET /crawl/{uuid}/status`` and returned by the three refresh calls. Attributes: enabled: Whether the crawl re-scrapes itself on a period. interval_seconds: Period between runs, 0 when disabled. status: ``DISABLED``, ``SCHEDULED``, ``RUNNING`` or ``FAILED``. generation: Number of refresh runs completed so far. last_run_at: ISO-8601 timestamp of the last completed run. next_run_at: ISO-8601 timestamp of the next due run, ``None`` when disabled. started_at: ISO-8601 start of the run in flight, ``None`` unless ``status`` is ``RUNNING``. Rendered by ``GET /crawl/{uuid}/status`` only. consecutive_failures: Failed runs since the last success, back to 0 on any success. Same status-only route as ``started_at``. error: Failure reason when ``status`` is ``FAILED``. history: Newest-last timeline, capped at the 50 most recent runs. """ def __init__(self, response_data: Dict[str, Any]): self._data = response_data # The three refresh endpoints answer with the state at the top level; # GET /status nests it under "refresh". One lookup, so the isinstance # narrowing holds for every read below it. nested = response_data.get('refresh') block = nested if isinstance(nested, dict) else response_data self.enabled: bool = bool(block.get('enabled', False)) self.interval_seconds: int = int(block.get('interval_seconds') or 0) self.status: str = block.get('status') or 'DISABLED' self.generation: int = int(block.get('generation') or 0) self.last_run_at: Optional[str] = block.get('last_run_at') self.next_run_at: Optional[str] = block.get('next_run_at') # ``GET /crawl/{uuid}/status`` relays the engine's refresh block # verbatim; the three refresh calls render a typed block that declares # neither of these, so both have to survive their absence. self.started_at: Optional[str] = block.get('started_at') self.consecutive_failures: int = int(block.get('consecutive_failures') or 0) self.error: Optional[str] = block.get('error') self.history: List[CrawlerRefreshEntry] = [ CrawlerRefreshEntry.from_dict(e) for e in (block.get('history') or []) ] @property def is_running(self) -> bool: """Whether a refresh run is in flight right now.""" return self.status == 'RUNNING' @property def last_run(self) -> Optional[CrawlerRefreshEntry]: """Most recent timeline row, ``None`` before the first run.""" return self.history[-1] if self.history else None def __len__(self) -> int: return len(self.history) def __iter__(self) -> Iterator[CrawlerRefreshEntry]: return iter(self.history) def __repr__(self): return ( f"CrawlerRefreshState(enabled={self.enabled}, status={self.status}, " f"interval_seconds={self.interval_seconds}, generation={self.generation}, " f"next_run_at={self.next_run_at!r})" )The
refreshblock of a crawl, as carried byGET /crawl/{uuid}/statusand returned by the three refresh calls.Attributes
enabled- Whether the crawl re-scrapes itself on a period.
interval_seconds- Period between runs, 0 when disabled.
statusDISABLED,SCHEDULED,RUNNINGorFAILED.generation- Number of refresh runs completed so far.
last_run_at- ISO-8601 timestamp of the last completed run.
next_run_at- ISO-8601 timestamp of the next due run,
Nonewhen disabled. started_at- ISO-8601 start of the run in flight,
NoneunlessstatusisRUNNING. Rendered byGET /crawl/{uuid}/statusonly. consecutive_failures- Failed runs since the last success, back to 0 on
any success. Same status-only route as
started_at. error- Failure reason when
statusisFAILED. history- Newest-last timeline, capped at the 50 most recent runs.
Instance variables
prop is_running : bool-
Expand source code
@property def is_running(self) -> bool: """Whether a refresh run is in flight right now.""" return self.status == 'RUNNING'Whether a refresh run is in flight right now.
prop last_run : CrawlerRefreshEntry | None-
Expand source code
@property def last_run(self) -> Optional[CrawlerRefreshEntry]: """Most recent timeline row, ``None`` before the first run.""" return self.history[-1] if self.history else NoneMost recent timeline row,
Nonebefore the first run.
class CrawlerScrapeResult (status_code: int,
country: str,
log_uuid: str,
log_url: str,
content: Dict[str, Any])-
Expand source code
@dataclass class CrawlerScrapeResult: """ The ``scrape`` sub-object of a ``crawler_url_visited`` payload. Attributes: status_code: HTTP status code returned by the target URL. country: 2-letter country code of the proxy that performed the scrape. log_uuid: ULID of the scrape log (used to fetch the full log later). log_url: Human-browseable dashboard URL for the log. content: Map of requested content format (``html``, ``text``, ``markdown``, ``clean_html``, ``json``, etc.) to the actual rendered string. The keys depend on what the caller requested in ``content_formats``. """ status_code: int country: str log_uuid: str log_url: str content: Dict[str, Any] @classmethod def from_dict(cls, data: Dict[str, Any]) -> 'CrawlerScrapeResult': return cls( status_code=data['status_code'], country=data['country'], log_uuid=data['log_uuid'], log_url=data['log_url'], content=data['content'], )The
scrapesub-object of acrawler_url_visitedpayload.Attributes
status_code- HTTP status code returned by the target URL.
country- 2-letter country code of the proxy that performed the scrape.
log_uuid- ULID of the scrape log (used to fetch the full log later).
log_url- Human-browseable dashboard URL for the log.
content- Map of requested content format (
html,text,markdown,clean_html,json, etc.) to the actual rendered string. The keys depend on what the caller requested incontent_formats.
Static methods
def from_dict(data: Dict[str, Any]) ‑> CrawlerScrapeResult
Instance variables
var content : Dict[str, Any]var country : strvar log_url : strvar log_uuid : strvar status_code : int
class CrawlerSearchCrawl (crawler_uuid: str,
documents: int | None = None,
vectors: int | None = None,
index: str | None = None)-
Expand source code
@dataclass class CrawlerSearchCrawl: """A crawl that was actually opened and searched.""" crawler_uuid: str documents: Optional[int] = None vectors: Optional[int] = None index: Optional[str] = None @classmethod def from_dict(cls, data: Dict[str, Any]) -> 'CrawlerSearchCrawl': return cls( crawler_uuid=data['crawler_uuid'], documents=data.get('documents'), vectors=data.get('vectors'), index=data.get('index'), )A crawl that was actually opened and searched.
Static methods
def from_dict(data: Dict[str, Any]) ‑> CrawlerSearchCrawl
Instance variables
var crawler_uuid : strvar documents : int | Nonevar index : str | Nonevar vectors : int | None
class CrawlerSearchError (message: str,
code: str,
http_status_code: int,
resource: str | None = None,
is_retryable: bool = False,
retry_delay: int | None = None,
retry_times: int | None = None,
documentation_url: str | None = None,
api_response: ForwardRef('ApiResponse') | None = None)-
Expand source code
class CrawlerSearchError(CrawlerError): """ Exception raised when ``POST /crawl/search`` cannot answer. Carries the API code, e.g. ``ERR::CRAWLER::SEARCH_NOT_ENABLED``, ``ERR::CRAWLER::SEARCH_NOT_READY``, ``ERR::CRAWLER::SEARCH_TOO_MANY_CRAWLS``. A crawl that is merely skipped is *not* an error: it is reported in ``CrawlerSearchResponse.skipped`` and the search still answers. """ passException raised when
POST /crawl/searchcannot answer.Carries the API code, e.g.
ERR::CRAWLER::SEARCH_NOT_ENABLED,ERR::CRAWLER::SEARCH_NOT_READY,ERR::CRAWLER::SEARCH_TOO_MANY_CRAWLS. A crawl that is merely skipped is not an error: it is reported inCrawlerSearchResponse.skippedand the search still answers.Ancestors
- CrawlerError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class CrawlerSearchResponse (response_data: Dict[str, Any])-
Expand source code
class CrawlerSearchResponse: """ Response from ``POST /crawl/search``. Returned by :py:meth:`ScrapflyClient.crawl_search`. The envelope states its own completeness: ``completeness == 'exact'`` with most crawls unopened is the normal outcome, because the fan-out proves via an admissible bound that the unopened crawls held nothing better. ``'partial'`` means the deadline cut the fan-out short. Attributes: query: The query as the server understood it. mode: ``vector``, ``fts`` or ``hybrid``. limit: The effective result cap. completeness: ``exact`` or ``partial``. results: Ranked :class:`CrawlerSearchResult` list. crawls: Crawls that were opened and searched. skipped: Requested crawls that contributed nothing, with a reason. stats: Timing/IO counters (``duration_ms``, ``crawls_searched``, ``candidates``, ``gcs_gets``). crawls_skipped_deadline: Crawler UUIDs the deadline cut before their leg ran. crawls_failed: Crawls whose leg errored, as :class:`CrawlerSearchSkipped` rows. cursor: Opaque token for the next page, ``None`` on the last page. Paging is cursor-based: an offset over a partial fan-out would re-run the legs and shift ranks. """ def __init__(self, response_data: Dict[str, Any]): self._data = response_data self.query: str = response_data['query'] self.mode: str = response_data['mode'] self.limit: int = response_data['limit'] self.completeness: str = response_data['completeness'] self.results: List[CrawlerSearchResult] = [ CrawlerSearchResult.from_dict(r) for r in response_data['results'] ] self.crawls: List[CrawlerSearchCrawl] = [ CrawlerSearchCrawl.from_dict(c) for c in (response_data.get('crawls') or []) ] self.skipped: List[CrawlerSearchSkipped] = [ CrawlerSearchSkipped.from_dict(s) for s in (response_data.get('skipped') or []) ] self.stats: Dict[str, Any] = response_data.get('stats') or {} self.crawls_requested: Optional[int] = response_data.get('crawls_requested') self.crawls_searched: Optional[int] = response_data.get('crawls_searched') self.crawls_pruned_exact: Optional[int] = response_data.get('crawls_pruned_exact') # These two name the crawls, they do not count them: a caller told # "3 failed" cannot act on it, and the crawls it can retry are the ones # the deadline cut. self.crawls_skipped_deadline: List[str] = list( response_data.get('crawls_skipped_deadline') or [] ) self.crawls_failed: List[CrawlerSearchSkipped] = [ CrawlerSearchSkipped.from_dict(c) for c in (response_data.get('crawls_failed') or []) ] self.theta: Optional[float] = response_data.get('theta') self.max_ub_unsearched: Optional[float] = response_data.get('max_ub_unsearched') self.cursor: Optional[str] = response_data.get('cursor') @property def is_exact(self) -> bool: """Whether the ranking is provably complete for the requested crawls.""" return self.completeness == 'exact' def __len__(self) -> int: return len(self.results) def __iter__(self) -> Iterator[CrawlerSearchResult]: return iter(self.results) def __repr__(self): return ( f"CrawlerSearchResponse(query={self.query!r}, mode={self.mode}, " f"results={len(self.results)}, completeness={self.completeness}, " f"skipped={len(self.skipped)})" )Response from
POST /crawl/search.Returned by :py:meth:
ScrapflyClient.crawl_search().The envelope states its own completeness:
completeness == 'exact'with most crawls unopened is the normal outcome, because the fan-out proves via an admissible bound that the unopened crawls held nothing better.'partial'means the deadline cut the fan-out short.Attributes
query- The query as the server understood it.
modevector,ftsorhybrid.limit- The effective result cap.
completenessexactorpartial.results- Ranked :class:
CrawlerSearchResultlist. crawls- Crawls that were opened and searched.
skipped- Requested crawls that contributed nothing, with a reason.
stats- Timing/IO counters (
duration_ms,crawls_searched,candidates,gcs_gets). crawls_skipped_deadline- Crawler UUIDs the deadline cut before their leg ran.
crawls_failed- Crawls whose leg errored, as
:class:
CrawlerSearchSkippedrows. cursor- Opaque token for the next page,
Noneon the last page. Paging is cursor-based: an offset over a partial fan-out would re-run the legs and shift ranks.
Instance variables
prop is_exact : bool-
Expand source code
@property def is_exact(self) -> bool: """Whether the ranking is provably complete for the requested crawls.""" return self.completeness == 'exact'Whether the ranking is provably complete for the requested crawls.
class CrawlerSearchResult (rank: int,
score: float,
crawler_uuid: str,
url: str,
chunk_id: int,
text: str,
scores: Dict[str, float] = <factory>,
title: str | None = None,
source_format: str | None = None,
content_type: str | None = None,
warc_offset: int | None = None,
warc_end: int | None = None,
contents_url: str | None = None)-
Expand source code
@dataclass class CrawlerSearchResult: """ One matched chunk from ``POST /crawl/search``. A result is a *chunk*, not a page: ``chunk_id`` orders chunks within one crawled document and ``text`` is only the matched slice. Use ``contents_url`` (or ``warc_offset``/``warc_end``) to expand a hit back to the full document. Attributes: rank: 1-based position in the merged ranking. score: The ranking score used for ordering (RRF in hybrid mode). scores: Per-leg scores (``vector``, ``fts``, ``rrf``); which keys are present depends on the mode. crawler_uuid: The crawl this chunk came from. url: The crawled URL. title: Document title, ``None`` when the page had none. source_format: Which stored format was indexed (``markdown``, ``text``, ``clean_html``, ``html``). content_type: Content type of the stored document. chunk_id: Chunk index within the document. text: The matched chunk text. warc_offset: Byte offset of the document record in the crawl WARC. warc_end: End byte offset of that record. contents_url: Ready-made ``/crawl/{uuid}/contents`` URL for the document this chunk belongs to. """ rank: int score: float crawler_uuid: str url: str chunk_id: int text: str scores: Dict[str, float] = field(default_factory=dict) title: Optional[str] = None source_format: Optional[str] = None content_type: Optional[str] = None warc_offset: Optional[int] = None warc_end: Optional[int] = None contents_url: Optional[str] = None @classmethod def from_dict(cls, data: Dict[str, Any]) -> 'CrawlerSearchResult': return cls( rank=data['rank'], score=data['score'], crawler_uuid=data['crawler_uuid'], url=data['url'], chunk_id=data['chunk_id'], text=data['text'], scores=data.get('scores') or {}, title=data.get('title'), source_format=data.get('source_format'), content_type=data.get('content_type'), warc_offset=data.get('warc_offset'), warc_end=data.get('warc_end'), contents_url=data.get('contents_url'), )One matched chunk from
POST /crawl/search.A result is a chunk, not a page:
chunk_idorders chunks within one crawled document andtextis only the matched slice. Usecontents_url(orwarc_offset/warc_end) to expand a hit back to the full document.Attributes
rank- 1-based position in the merged ranking.
score- The ranking score used for ordering (RRF in hybrid mode).
scores- Per-leg scores (
vector,fts,rrf); which keys are present depends on the mode. crawler_uuid- The crawl this chunk came from.
url- The crawled URL.
title- Document title,
Nonewhen the page had none. source_format- Which stored format was indexed (
markdown,text,clean_html,html). content_type- Content type of the stored document.
chunk_id- Chunk index within the document.
text- The matched chunk text.
warc_offset- Byte offset of the document record in the crawl WARC.
warc_end- End byte offset of that record.
contents_url- Ready-made
/crawl/{uuid}/contentsURL for the document this chunk belongs to.
Static methods
def from_dict(data: Dict[str, Any]) ‑> CrawlerSearchResult
Instance variables
var chunk_id : intvar content_type : str | Nonevar contents_url : str | Nonevar crawler_uuid : strvar rank : intvar score : floatvar scores : Dict[str, float]var source_format : str | Nonevar text : strvar title : str | Nonevar url : strvar warc_end : int | Nonevar warc_offset : int | None
class CrawlerSearchSkipped (crawler_uuid: str, reason: str, status: str | None = None)-
Expand source code
@dataclass class CrawlerSearchSkipped: """ A requested crawl that contributed nothing, and why. Skips are never fatal: the search still answers with whatever the other crawls returned. ``reason`` is one of ``search_not_enabled``, ``search_not_ready``, ``search_failed``, ``search_disabled``, ``incompatible_index``, ``deadline``. """ crawler_uuid: str reason: str status: Optional[str] = None @classmethod def from_dict(cls, data: Dict[str, Any]) -> 'CrawlerSearchSkipped': return cls( crawler_uuid=data['crawler_uuid'], reason=data['reason'], status=data.get('status'), )A requested crawl that contributed nothing, and why.
Skips are never fatal: the search still answers with whatever the other crawls returned.
reasonis one ofsearch_not_enabled,search_not_ready,search_failed,search_disabled,incompatible_index,deadline.Static methods
def from_dict(data: Dict[str, Any]) ‑> CrawlerSearchSkipped
Instance variables
var crawler_uuid : strvar reason : strvar status : str | None
class CrawlerSearchState (status: str,
manifest: str | None = None,
documents: int | None = None,
vectors: int | None = None,
dropped: int | None = None,
queue_depth: int | None = None,
fragments: int | None = None,
error: str | None = None,
built_at: str | None = None,
index: str | None = None,
generation: int | None = None)-
Expand source code
@dataclass class CrawlerSearchState: """ The ``search`` block describing a crawl's index, as carried by ``GET /crawl/{uuid}/status`` and by the two search webhooks. Attributes: status: ``DISABLED``, ``BUILDING``, ``READY``, ``PARTIAL`` or ``FAILED``. Only ``READY`` and ``PARTIAL`` are searchable. manifest: Storage path of the index manifest, ``None`` until the artifact is published. documents: Crawled documents represented in the index. vectors: Embedded chunks. dropped: Chunks discarded during the build (embedding failures, oversized rows). queue_depth: Chunks still waiting to be embedded at snapshot time. fragments: Published Lance fragments. error: Failure reason when ``status`` is ``FAILED``. built_at: ISO-8601 timestamp of the terminal publish. index: Vector index type (e.g. ``IVF_PQ``), ``None`` when the row count stayed below the index threshold. generation: Build generation, bumped when a paused crawl resumes and rebuilds. Results from different generations are not comparable. """ status: str manifest: Optional[str] = None documents: Optional[int] = None vectors: Optional[int] = None dropped: Optional[int] = None queue_depth: Optional[int] = None fragments: Optional[int] = None error: Optional[str] = None built_at: Optional[str] = None index: Optional[str] = None generation: Optional[int] = None @classmethod def from_dict(cls, data: Dict[str, Any]) -> 'CrawlerSearchState': return cls( status=data['status'], manifest=data.get('manifest'), documents=data.get('documents'), vectors=data.get('vectors'), dropped=data.get('dropped'), queue_depth=data.get('queue_depth'), fragments=data.get('fragments'), error=data.get('error'), built_at=data.get('built_at'), index=data.get('index'), generation=data.get('generation'), ) @property def is_searchable(self) -> bool: """Whether the index can answer a query right now.""" return self.status in ('READY', 'PARTIAL')The
searchblock describing a crawl's index, as carried byGET /crawl/{uuid}/statusand by the two search webhooks.Attributes
statusDISABLED,BUILDING,READY,PARTIALorFAILED. OnlyREADYandPARTIALare searchable.manifest- Storage path of the index manifest,
Noneuntil the artifact is published. documents- Crawled documents represented in the index.
vectors- Embedded chunks.
dropped- Chunks discarded during the build (embedding failures, oversized rows).
queue_depth- Chunks still waiting to be embedded at snapshot time.
fragments- Published Lance fragments.
error- Failure reason when
statusisFAILED. built_at- ISO-8601 timestamp of the terminal publish.
index- Vector index type (e.g.
IVF_PQ),Nonewhen the row count stayed below the index threshold. generation- Build generation, bumped when a paused crawl resumes and rebuilds. Results from different generations are not comparable.
Static methods
def from_dict(data: Dict[str, Any]) ‑> CrawlerSearchState
Instance variables
var built_at : str | Nonevar documents : int | Nonevar dropped : int | Nonevar error : str | Nonevar fragments : int | Nonevar generation : int | Nonevar index : str | Noneprop is_searchable : bool-
Expand source code
@property def is_searchable(self) -> bool: """Whether the index can answer a query right now.""" return self.status in ('READY', 'PARTIAL')Whether the index can answer a query right now.
var manifest : str | Nonevar queue_depth : int | Nonevar status : strvar vectors : int | None
class CrawlerSearchWebhook (event: str,
crawler_uuid: str,
project: str,
env: str,
action: str,
state: CrawlerState,
seed_url: str,
status_link: str,
search: CrawlerSearchState)-
Expand source code
@dataclass class CrawlerSearchWebhook(CrawlerWebhookBase): """ Payload for ``crawler_search_ready`` and ``crawler_search_failed``. The search index is published after the crawl's own success classification and can fail without the crawl failing, so these events are emitted separately from the lifecycle ones. Disambiguate on ``self.event`` or on ``self.search.status``. Attributes: seed_url: The root URL the crawl was started from. status_link: URL to fetch the live crawler status. search: The index state block. """ seed_url: str status_link: str search: CrawlerSearchState @classmethod def from_payload(cls, event: str, payload: Dict[str, Any]) -> 'CrawlerSearchWebhook': # Not _parse_base: the two search events are the only ones Scrapfly # emits without an `action` tag, so requiring it would reject every # valid payload. return cls( event=event, crawler_uuid=payload['crawler_uuid'], project=payload['project'], env=payload['env'], action=payload.get('action', ''), state=CrawlerState(payload['state']), seed_url=payload['seed_url'], status_link=payload['links']['status'], search=CrawlerSearchState.from_dict(payload['search']), )Payload for
crawler_search_readyandcrawler_search_failed.The search index is published after the crawl's own success classification and can fail without the crawl failing, so these events are emitted separately from the lifecycle ones. Disambiguate on
self.eventor onself.search.status.Attributes
seed_url- The root URL the crawl was started from.
status_link- URL to fetch the live crawler status.
search- The index state block.
Ancestors
Static methods
def from_payload(event: str, payload: Dict[str, Any]) ‑> CrawlerSearchWebhook
Instance variables
var search : CrawlerSearchStatevar seed_url : strvar status_link : str
class CrawlerStartResponse (response_data: Dict[str, Any])-
Expand source code
class CrawlerStartResponse: """ Response from starting a crawler job Returned by ScrapflyClient.start_crawl() method. Strict parsing: ``uuid`` and ``status`` are part of the documented contract and are required. A missing field raises ``KeyError`` so the caller knows immediately that the API contract changed. Attributes: uuid: Unique identifier for the crawler job status: Initial status (typically 'PENDING') """ def __init__(self, response_data: Dict[str, Any]): """ Initialize from API response Args: response_data: Raw API response dictionary """ self._data = response_data # API canonical name is `crawler_uuid`; we accept `uuid` only as a # legacy fallback, in case an older server emits the short form. if 'crawler_uuid' in response_data: self.uuid = response_data['crawler_uuid'] elif 'uuid' in response_data: self.uuid = response_data['uuid'] else: raise KeyError( "CrawlerStartResponse: required field 'crawler_uuid' (or legacy 'uuid') is missing" ) self.status = response_data['status'] assert isinstance(self.uuid, str) and self.uuid, ( f"CrawlerStartResponse: uuid must be a non-empty string, got {self.uuid!r}" ) assert isinstance(self.status, str) and self.status, ( f"CrawlerStartResponse: status must be a non-empty string, got {self.status!r}" ) def __repr__(self): return f"CrawlerStartResponse(uuid={self.uuid}, status={self.status})"Response from starting a crawler job
Returned by ScrapflyClient.start_crawl() method.
Strict parsing:
uuidandstatusare part of the documented contract and are required. A missing field raisesKeyErrorso the caller knows immediately that the API contract changed.Attributes
uuid- Unique identifier for the crawler job
status- Initial status (typically 'PENDING')
Initialize from API response
Args
response_data- Raw API response dictionary
class CrawlerState (state: Dict[str, Any])-
Expand source code
class CrawlerState: """ Nested ``state`` block of a crawler status response. Field names match the wire format emitted by Scrapfly, which is the single source of truth. Go and TypeScript SDKs expose the same names on their ``status.state`` object. Attributes: urls_visited: Number of URLs successfully crawled. urls_extracted: Total URLs discovered (seed + links + sitemaps). urls_to_crawl: Derived as ``urls_extracted - urls_skipped`` server-side. urls_failed: URLs that failed to crawl. urls_skipped: URLs skipped (filtered by exclude rules, robots.txt, etc.). api_credit_used: Total API credits consumed by this crawl. duration: Elapsed time in seconds. start_time: Unix epoch seconds when the first worker picked up the job, or ``None`` while the job is still in ``PENDING``. stop_time: Unix epoch seconds when the crawler reached a terminal state, or ``None`` while still running. stop_reason: Reason for stop (``page_limit``, ``max_duration``, etc.), or ``None`` while still running. """ __slots__ = ( 'urls_visited', 'urls_extracted', 'urls_to_crawl', 'urls_failed', 'urls_skipped', 'api_credit_used', 'duration', 'start_time', 'stop_time', 'stop_reason', ) def __init__(self, state: Dict[str, Any]): assert isinstance(state, dict), ( f"CrawlerState: expected dict, got {type(state).__name__}" ) self.urls_visited: int = state['urls_visited'] self.urls_extracted: int = state['urls_extracted'] self.urls_to_crawl: int = state['urls_to_crawl'] self.urls_failed: int = state['urls_failed'] self.urls_skipped: int = state['urls_skipped'] self.api_credit_used = state['api_credit_used'] self.duration = state['duration'] # Nullable during PENDING — before a worker has picked up the job. self.start_time: Optional[int] = state.get('start_time') self.stop_time: Optional[int] = state.get('stop_time') self.stop_reason: Optional[str] = state.get('stop_reason') def __repr__(self): return ( f"CrawlerState(visited={self.urls_visited}, extracted={self.urls_extracted}, " f"to_crawl={self.urls_to_crawl}, failed={self.urls_failed}, " f"skipped={self.urls_skipped})" )Nested
stateblock of a crawler status response.Field names match the wire format emitted by Scrapfly, which is the single source of truth. Go and TypeScript SDKs expose the same names on their
status.stateobject.Attributes
urls_visited- Number of URLs successfully crawled.
urls_extracted- Total URLs discovered (seed + links + sitemaps).
urls_to_crawl- Derived as
urls_extracted - urls_skippedserver-side. urls_failed- URLs that failed to crawl.
urls_skipped- URLs skipped (filtered by exclude rules, robots.txt, etc.).
api_credit_used- Total API credits consumed by this crawl.
duration- Elapsed time in seconds.
start_time- Unix epoch seconds when the first worker picked up the job,
or
Nonewhile the job is still inPENDING. stop_time- Unix epoch seconds when the crawler reached a terminal state,
or
Nonewhile still running. stop_reason- Reason for stop (
page_limit,max_duration, etc.), orNonewhile still running.
Instance variables
var api_credit_used-
Expand source code
class CrawlerState: """ Nested ``state`` block of a crawler status response. Field names match the wire format emitted by Scrapfly, which is the single source of truth. Go and TypeScript SDKs expose the same names on their ``status.state`` object. Attributes: urls_visited: Number of URLs successfully crawled. urls_extracted: Total URLs discovered (seed + links + sitemaps). urls_to_crawl: Derived as ``urls_extracted - urls_skipped`` server-side. urls_failed: URLs that failed to crawl. urls_skipped: URLs skipped (filtered by exclude rules, robots.txt, etc.). api_credit_used: Total API credits consumed by this crawl. duration: Elapsed time in seconds. start_time: Unix epoch seconds when the first worker picked up the job, or ``None`` while the job is still in ``PENDING``. stop_time: Unix epoch seconds when the crawler reached a terminal state, or ``None`` while still running. stop_reason: Reason for stop (``page_limit``, ``max_duration``, etc.), or ``None`` while still running. """ __slots__ = ( 'urls_visited', 'urls_extracted', 'urls_to_crawl', 'urls_failed', 'urls_skipped', 'api_credit_used', 'duration', 'start_time', 'stop_time', 'stop_reason', ) def __init__(self, state: Dict[str, Any]): assert isinstance(state, dict), ( f"CrawlerState: expected dict, got {type(state).__name__}" ) self.urls_visited: int = state['urls_visited'] self.urls_extracted: int = state['urls_extracted'] self.urls_to_crawl: int = state['urls_to_crawl'] self.urls_failed: int = state['urls_failed'] self.urls_skipped: int = state['urls_skipped'] self.api_credit_used = state['api_credit_used'] self.duration = state['duration'] # Nullable during PENDING — before a worker has picked up the job. self.start_time: Optional[int] = state.get('start_time') self.stop_time: Optional[int] = state.get('stop_time') self.stop_reason: Optional[str] = state.get('stop_reason') def __repr__(self): return ( f"CrawlerState(visited={self.urls_visited}, extracted={self.urls_extracted}, " f"to_crawl={self.urls_to_crawl}, failed={self.urls_failed}, " f"skipped={self.urls_skipped})" ) var duration-
Expand source code
class CrawlerState: """ Nested ``state`` block of a crawler status response. Field names match the wire format emitted by Scrapfly, which is the single source of truth. Go and TypeScript SDKs expose the same names on their ``status.state`` object. Attributes: urls_visited: Number of URLs successfully crawled. urls_extracted: Total URLs discovered (seed + links + sitemaps). urls_to_crawl: Derived as ``urls_extracted - urls_skipped`` server-side. urls_failed: URLs that failed to crawl. urls_skipped: URLs skipped (filtered by exclude rules, robots.txt, etc.). api_credit_used: Total API credits consumed by this crawl. duration: Elapsed time in seconds. start_time: Unix epoch seconds when the first worker picked up the job, or ``None`` while the job is still in ``PENDING``. stop_time: Unix epoch seconds when the crawler reached a terminal state, or ``None`` while still running. stop_reason: Reason for stop (``page_limit``, ``max_duration``, etc.), or ``None`` while still running. """ __slots__ = ( 'urls_visited', 'urls_extracted', 'urls_to_crawl', 'urls_failed', 'urls_skipped', 'api_credit_used', 'duration', 'start_time', 'stop_time', 'stop_reason', ) def __init__(self, state: Dict[str, Any]): assert isinstance(state, dict), ( f"CrawlerState: expected dict, got {type(state).__name__}" ) self.urls_visited: int = state['urls_visited'] self.urls_extracted: int = state['urls_extracted'] self.urls_to_crawl: int = state['urls_to_crawl'] self.urls_failed: int = state['urls_failed'] self.urls_skipped: int = state['urls_skipped'] self.api_credit_used = state['api_credit_used'] self.duration = state['duration'] # Nullable during PENDING — before a worker has picked up the job. self.start_time: Optional[int] = state.get('start_time') self.stop_time: Optional[int] = state.get('stop_time') self.stop_reason: Optional[str] = state.get('stop_reason') def __repr__(self): return ( f"CrawlerState(visited={self.urls_visited}, extracted={self.urls_extracted}, " f"to_crawl={self.urls_to_crawl}, failed={self.urls_failed}, " f"skipped={self.urls_skipped})" ) var start_time-
Expand source code
class CrawlerState: """ Nested ``state`` block of a crawler status response. Field names match the wire format emitted by Scrapfly, which is the single source of truth. Go and TypeScript SDKs expose the same names on their ``status.state`` object. Attributes: urls_visited: Number of URLs successfully crawled. urls_extracted: Total URLs discovered (seed + links + sitemaps). urls_to_crawl: Derived as ``urls_extracted - urls_skipped`` server-side. urls_failed: URLs that failed to crawl. urls_skipped: URLs skipped (filtered by exclude rules, robots.txt, etc.). api_credit_used: Total API credits consumed by this crawl. duration: Elapsed time in seconds. start_time: Unix epoch seconds when the first worker picked up the job, or ``None`` while the job is still in ``PENDING``. stop_time: Unix epoch seconds when the crawler reached a terminal state, or ``None`` while still running. stop_reason: Reason for stop (``page_limit``, ``max_duration``, etc.), or ``None`` while still running. """ __slots__ = ( 'urls_visited', 'urls_extracted', 'urls_to_crawl', 'urls_failed', 'urls_skipped', 'api_credit_used', 'duration', 'start_time', 'stop_time', 'stop_reason', ) def __init__(self, state: Dict[str, Any]): assert isinstance(state, dict), ( f"CrawlerState: expected dict, got {type(state).__name__}" ) self.urls_visited: int = state['urls_visited'] self.urls_extracted: int = state['urls_extracted'] self.urls_to_crawl: int = state['urls_to_crawl'] self.urls_failed: int = state['urls_failed'] self.urls_skipped: int = state['urls_skipped'] self.api_credit_used = state['api_credit_used'] self.duration = state['duration'] # Nullable during PENDING — before a worker has picked up the job. self.start_time: Optional[int] = state.get('start_time') self.stop_time: Optional[int] = state.get('stop_time') self.stop_reason: Optional[str] = state.get('stop_reason') def __repr__(self): return ( f"CrawlerState(visited={self.urls_visited}, extracted={self.urls_extracted}, " f"to_crawl={self.urls_to_crawl}, failed={self.urls_failed}, " f"skipped={self.urls_skipped})" ) var stop_reason-
Expand source code
class CrawlerState: """ Nested ``state`` block of a crawler status response. Field names match the wire format emitted by Scrapfly, which is the single source of truth. Go and TypeScript SDKs expose the same names on their ``status.state`` object. Attributes: urls_visited: Number of URLs successfully crawled. urls_extracted: Total URLs discovered (seed + links + sitemaps). urls_to_crawl: Derived as ``urls_extracted - urls_skipped`` server-side. urls_failed: URLs that failed to crawl. urls_skipped: URLs skipped (filtered by exclude rules, robots.txt, etc.). api_credit_used: Total API credits consumed by this crawl. duration: Elapsed time in seconds. start_time: Unix epoch seconds when the first worker picked up the job, or ``None`` while the job is still in ``PENDING``. stop_time: Unix epoch seconds when the crawler reached a terminal state, or ``None`` while still running. stop_reason: Reason for stop (``page_limit``, ``max_duration``, etc.), or ``None`` while still running. """ __slots__ = ( 'urls_visited', 'urls_extracted', 'urls_to_crawl', 'urls_failed', 'urls_skipped', 'api_credit_used', 'duration', 'start_time', 'stop_time', 'stop_reason', ) def __init__(self, state: Dict[str, Any]): assert isinstance(state, dict), ( f"CrawlerState: expected dict, got {type(state).__name__}" ) self.urls_visited: int = state['urls_visited'] self.urls_extracted: int = state['urls_extracted'] self.urls_to_crawl: int = state['urls_to_crawl'] self.urls_failed: int = state['urls_failed'] self.urls_skipped: int = state['urls_skipped'] self.api_credit_used = state['api_credit_used'] self.duration = state['duration'] # Nullable during PENDING — before a worker has picked up the job. self.start_time: Optional[int] = state.get('start_time') self.stop_time: Optional[int] = state.get('stop_time') self.stop_reason: Optional[str] = state.get('stop_reason') def __repr__(self): return ( f"CrawlerState(visited={self.urls_visited}, extracted={self.urls_extracted}, " f"to_crawl={self.urls_to_crawl}, failed={self.urls_failed}, " f"skipped={self.urls_skipped})" ) var stop_time-
Expand source code
class CrawlerState: """ Nested ``state`` block of a crawler status response. Field names match the wire format emitted by Scrapfly, which is the single source of truth. Go and TypeScript SDKs expose the same names on their ``status.state`` object. Attributes: urls_visited: Number of URLs successfully crawled. urls_extracted: Total URLs discovered (seed + links + sitemaps). urls_to_crawl: Derived as ``urls_extracted - urls_skipped`` server-side. urls_failed: URLs that failed to crawl. urls_skipped: URLs skipped (filtered by exclude rules, robots.txt, etc.). api_credit_used: Total API credits consumed by this crawl. duration: Elapsed time in seconds. start_time: Unix epoch seconds when the first worker picked up the job, or ``None`` while the job is still in ``PENDING``. stop_time: Unix epoch seconds when the crawler reached a terminal state, or ``None`` while still running. stop_reason: Reason for stop (``page_limit``, ``max_duration``, etc.), or ``None`` while still running. """ __slots__ = ( 'urls_visited', 'urls_extracted', 'urls_to_crawl', 'urls_failed', 'urls_skipped', 'api_credit_used', 'duration', 'start_time', 'stop_time', 'stop_reason', ) def __init__(self, state: Dict[str, Any]): assert isinstance(state, dict), ( f"CrawlerState: expected dict, got {type(state).__name__}" ) self.urls_visited: int = state['urls_visited'] self.urls_extracted: int = state['urls_extracted'] self.urls_to_crawl: int = state['urls_to_crawl'] self.urls_failed: int = state['urls_failed'] self.urls_skipped: int = state['urls_skipped'] self.api_credit_used = state['api_credit_used'] self.duration = state['duration'] # Nullable during PENDING — before a worker has picked up the job. self.start_time: Optional[int] = state.get('start_time') self.stop_time: Optional[int] = state.get('stop_time') self.stop_reason: Optional[str] = state.get('stop_reason') def __repr__(self): return ( f"CrawlerState(visited={self.urls_visited}, extracted={self.urls_extracted}, " f"to_crawl={self.urls_to_crawl}, failed={self.urls_failed}, " f"skipped={self.urls_skipped})" ) var urls_extracted-
Expand source code
class CrawlerState: """ Nested ``state`` block of a crawler status response. Field names match the wire format emitted by Scrapfly, which is the single source of truth. Go and TypeScript SDKs expose the same names on their ``status.state`` object. Attributes: urls_visited: Number of URLs successfully crawled. urls_extracted: Total URLs discovered (seed + links + sitemaps). urls_to_crawl: Derived as ``urls_extracted - urls_skipped`` server-side. urls_failed: URLs that failed to crawl. urls_skipped: URLs skipped (filtered by exclude rules, robots.txt, etc.). api_credit_used: Total API credits consumed by this crawl. duration: Elapsed time in seconds. start_time: Unix epoch seconds when the first worker picked up the job, or ``None`` while the job is still in ``PENDING``. stop_time: Unix epoch seconds when the crawler reached a terminal state, or ``None`` while still running. stop_reason: Reason for stop (``page_limit``, ``max_duration``, etc.), or ``None`` while still running. """ __slots__ = ( 'urls_visited', 'urls_extracted', 'urls_to_crawl', 'urls_failed', 'urls_skipped', 'api_credit_used', 'duration', 'start_time', 'stop_time', 'stop_reason', ) def __init__(self, state: Dict[str, Any]): assert isinstance(state, dict), ( f"CrawlerState: expected dict, got {type(state).__name__}" ) self.urls_visited: int = state['urls_visited'] self.urls_extracted: int = state['urls_extracted'] self.urls_to_crawl: int = state['urls_to_crawl'] self.urls_failed: int = state['urls_failed'] self.urls_skipped: int = state['urls_skipped'] self.api_credit_used = state['api_credit_used'] self.duration = state['duration'] # Nullable during PENDING — before a worker has picked up the job. self.start_time: Optional[int] = state.get('start_time') self.stop_time: Optional[int] = state.get('stop_time') self.stop_reason: Optional[str] = state.get('stop_reason') def __repr__(self): return ( f"CrawlerState(visited={self.urls_visited}, extracted={self.urls_extracted}, " f"to_crawl={self.urls_to_crawl}, failed={self.urls_failed}, " f"skipped={self.urls_skipped})" ) var urls_failed-
Expand source code
class CrawlerState: """ Nested ``state`` block of a crawler status response. Field names match the wire format emitted by Scrapfly, which is the single source of truth. Go and TypeScript SDKs expose the same names on their ``status.state`` object. Attributes: urls_visited: Number of URLs successfully crawled. urls_extracted: Total URLs discovered (seed + links + sitemaps). urls_to_crawl: Derived as ``urls_extracted - urls_skipped`` server-side. urls_failed: URLs that failed to crawl. urls_skipped: URLs skipped (filtered by exclude rules, robots.txt, etc.). api_credit_used: Total API credits consumed by this crawl. duration: Elapsed time in seconds. start_time: Unix epoch seconds when the first worker picked up the job, or ``None`` while the job is still in ``PENDING``. stop_time: Unix epoch seconds when the crawler reached a terminal state, or ``None`` while still running. stop_reason: Reason for stop (``page_limit``, ``max_duration``, etc.), or ``None`` while still running. """ __slots__ = ( 'urls_visited', 'urls_extracted', 'urls_to_crawl', 'urls_failed', 'urls_skipped', 'api_credit_used', 'duration', 'start_time', 'stop_time', 'stop_reason', ) def __init__(self, state: Dict[str, Any]): assert isinstance(state, dict), ( f"CrawlerState: expected dict, got {type(state).__name__}" ) self.urls_visited: int = state['urls_visited'] self.urls_extracted: int = state['urls_extracted'] self.urls_to_crawl: int = state['urls_to_crawl'] self.urls_failed: int = state['urls_failed'] self.urls_skipped: int = state['urls_skipped'] self.api_credit_used = state['api_credit_used'] self.duration = state['duration'] # Nullable during PENDING — before a worker has picked up the job. self.start_time: Optional[int] = state.get('start_time') self.stop_time: Optional[int] = state.get('stop_time') self.stop_reason: Optional[str] = state.get('stop_reason') def __repr__(self): return ( f"CrawlerState(visited={self.urls_visited}, extracted={self.urls_extracted}, " f"to_crawl={self.urls_to_crawl}, failed={self.urls_failed}, " f"skipped={self.urls_skipped})" ) var urls_skipped-
Expand source code
class CrawlerState: """ Nested ``state`` block of a crawler status response. Field names match the wire format emitted by Scrapfly, which is the single source of truth. Go and TypeScript SDKs expose the same names on their ``status.state`` object. Attributes: urls_visited: Number of URLs successfully crawled. urls_extracted: Total URLs discovered (seed + links + sitemaps). urls_to_crawl: Derived as ``urls_extracted - urls_skipped`` server-side. urls_failed: URLs that failed to crawl. urls_skipped: URLs skipped (filtered by exclude rules, robots.txt, etc.). api_credit_used: Total API credits consumed by this crawl. duration: Elapsed time in seconds. start_time: Unix epoch seconds when the first worker picked up the job, or ``None`` while the job is still in ``PENDING``. stop_time: Unix epoch seconds when the crawler reached a terminal state, or ``None`` while still running. stop_reason: Reason for stop (``page_limit``, ``max_duration``, etc.), or ``None`` while still running. """ __slots__ = ( 'urls_visited', 'urls_extracted', 'urls_to_crawl', 'urls_failed', 'urls_skipped', 'api_credit_used', 'duration', 'start_time', 'stop_time', 'stop_reason', ) def __init__(self, state: Dict[str, Any]): assert isinstance(state, dict), ( f"CrawlerState: expected dict, got {type(state).__name__}" ) self.urls_visited: int = state['urls_visited'] self.urls_extracted: int = state['urls_extracted'] self.urls_to_crawl: int = state['urls_to_crawl'] self.urls_failed: int = state['urls_failed'] self.urls_skipped: int = state['urls_skipped'] self.api_credit_used = state['api_credit_used'] self.duration = state['duration'] # Nullable during PENDING — before a worker has picked up the job. self.start_time: Optional[int] = state.get('start_time') self.stop_time: Optional[int] = state.get('stop_time') self.stop_reason: Optional[str] = state.get('stop_reason') def __repr__(self): return ( f"CrawlerState(visited={self.urls_visited}, extracted={self.urls_extracted}, " f"to_crawl={self.urls_to_crawl}, failed={self.urls_failed}, " f"skipped={self.urls_skipped})" ) var urls_to_crawl-
Expand source code
class CrawlerState: """ Nested ``state`` block of a crawler status response. Field names match the wire format emitted by Scrapfly, which is the single source of truth. Go and TypeScript SDKs expose the same names on their ``status.state`` object. Attributes: urls_visited: Number of URLs successfully crawled. urls_extracted: Total URLs discovered (seed + links + sitemaps). urls_to_crawl: Derived as ``urls_extracted - urls_skipped`` server-side. urls_failed: URLs that failed to crawl. urls_skipped: URLs skipped (filtered by exclude rules, robots.txt, etc.). api_credit_used: Total API credits consumed by this crawl. duration: Elapsed time in seconds. start_time: Unix epoch seconds when the first worker picked up the job, or ``None`` while the job is still in ``PENDING``. stop_time: Unix epoch seconds when the crawler reached a terminal state, or ``None`` while still running. stop_reason: Reason for stop (``page_limit``, ``max_duration``, etc.), or ``None`` while still running. """ __slots__ = ( 'urls_visited', 'urls_extracted', 'urls_to_crawl', 'urls_failed', 'urls_skipped', 'api_credit_used', 'duration', 'start_time', 'stop_time', 'stop_reason', ) def __init__(self, state: Dict[str, Any]): assert isinstance(state, dict), ( f"CrawlerState: expected dict, got {type(state).__name__}" ) self.urls_visited: int = state['urls_visited'] self.urls_extracted: int = state['urls_extracted'] self.urls_to_crawl: int = state['urls_to_crawl'] self.urls_failed: int = state['urls_failed'] self.urls_skipped: int = state['urls_skipped'] self.api_credit_used = state['api_credit_used'] self.duration = state['duration'] # Nullable during PENDING — before a worker has picked up the job. self.start_time: Optional[int] = state.get('start_time') self.stop_time: Optional[int] = state.get('stop_time') self.stop_reason: Optional[str] = state.get('stop_reason') def __repr__(self): return ( f"CrawlerState(visited={self.urls_visited}, extracted={self.urls_extracted}, " f"to_crawl={self.urls_to_crawl}, failed={self.urls_failed}, " f"skipped={self.urls_skipped})" ) var urls_visited-
Expand source code
class CrawlerState: """ Nested ``state`` block of a crawler status response. Field names match the wire format emitted by Scrapfly, which is the single source of truth. Go and TypeScript SDKs expose the same names on their ``status.state`` object. Attributes: urls_visited: Number of URLs successfully crawled. urls_extracted: Total URLs discovered (seed + links + sitemaps). urls_to_crawl: Derived as ``urls_extracted - urls_skipped`` server-side. urls_failed: URLs that failed to crawl. urls_skipped: URLs skipped (filtered by exclude rules, robots.txt, etc.). api_credit_used: Total API credits consumed by this crawl. duration: Elapsed time in seconds. start_time: Unix epoch seconds when the first worker picked up the job, or ``None`` while the job is still in ``PENDING``. stop_time: Unix epoch seconds when the crawler reached a terminal state, or ``None`` while still running. stop_reason: Reason for stop (``page_limit``, ``max_duration``, etc.), or ``None`` while still running. """ __slots__ = ( 'urls_visited', 'urls_extracted', 'urls_to_crawl', 'urls_failed', 'urls_skipped', 'api_credit_used', 'duration', 'start_time', 'stop_time', 'stop_reason', ) def __init__(self, state: Dict[str, Any]): assert isinstance(state, dict), ( f"CrawlerState: expected dict, got {type(state).__name__}" ) self.urls_visited: int = state['urls_visited'] self.urls_extracted: int = state['urls_extracted'] self.urls_to_crawl: int = state['urls_to_crawl'] self.urls_failed: int = state['urls_failed'] self.urls_skipped: int = state['urls_skipped'] self.api_credit_used = state['api_credit_used'] self.duration = state['duration'] # Nullable during PENDING — before a worker has picked up the job. self.start_time: Optional[int] = state.get('start_time') self.stop_time: Optional[int] = state.get('stop_time') self.stop_reason: Optional[str] = state.get('stop_reason') def __repr__(self): return ( f"CrawlerState(visited={self.urls_visited}, extracted={self.urls_extracted}, " f"to_crawl={self.urls_to_crawl}, failed={self.urls_failed}, " f"skipped={self.urls_skipped})" )
class CrawlerStatusResponse (response_data: Dict[str, Any])-
Expand source code
class CrawlerStatusResponse: """ Response from checking crawler job status. Returned by :py:meth:`ScrapflyClient.get_crawl_status`. Provides real-time progress tracking for crawler jobs. **Field names match the wire format.** Scrapfly is the source of truth; the Go and TypeScript SDKs expose identical names. Access state counters via the nested ``state`` attribute: >>> status.state.urls_visited 12 >>> status.state.urls_extracted 34 Attributes: uuid: Crawler job UUID. status: Current status (``PENDING``, ``RUNNING``, ``DONE``, ``CANCELLED``). is_success: Whether the crawler job completed successfully (``None`` while running). is_finished: Whether the crawler job has finished (regardless of success/failure). state: :class:`CrawlerState` — all the per-crawl counters and timings. search: :class:`CrawlerSearchState` when the crawl was started with ``search=True``, otherwise ``None``. refresh: :class:`CrawlerRefreshState` when the crawl re-scrapes itself on a period, otherwise ``None``. """ # Status constants STATUS_PENDING = 'PENDING' STATUS_RUNNING = 'RUNNING' STATUS_DONE = 'DONE' STATUS_CANCELLED = 'CANCELLED' def __init__(self, response_data: Dict[str, Any]): """ Initialize from API response. Strict parsing: required fields (``crawler_uuid``, ``status``, ``is_success``, ``is_finished``, and the documented ``state.*`` metrics) are read with direct access so missing keys raise ``KeyError`` at parse time. This catches API contract drift loud and early. Args: response_data: Raw API response dictionary. """ self._data = response_data # Identification — accept legacy `uuid` only as fallback. if 'crawler_uuid' in response_data: self.uuid = response_data['crawler_uuid'] elif 'uuid' in response_data: self.uuid = response_data['uuid'] else: raise KeyError( "CrawlerStatusResponse: required field 'crawler_uuid' (or legacy 'uuid') is missing" ) self.status = response_data['status'] # `is_success` may legitimately be `null` while still running. self.is_success = response_data['is_success'] self.is_finished = response_data['is_finished'] assert isinstance(self.uuid, str) and self.uuid, ( f"CrawlerStatusResponse: uuid must be a non-empty string, got {self.uuid!r}" ) assert isinstance(self.status, str) and self.status, ( f"CrawlerStatusResponse: status must be a non-empty string, got {self.status!r}" ) assert isinstance(self.is_finished, bool), ( f"CrawlerStatusResponse: is_finished must be bool, got {type(self.is_finished).__name__}" ) assert self.is_success is None or isinstance(self.is_success, bool), ( f"CrawlerStatusResponse: is_success must be bool or None, got {type(self.is_success).__name__}" ) # Nested state — canonical shape matching Go / TS SDKs. self.state = CrawlerState(response_data['state']) # Search index state. Optional: only crawls started with search=True # carry the block, and older API builds omit it entirely. search = response_data.get('search') self.search: Optional[CrawlerSearchState] = ( CrawlerSearchState.from_dict(search) if search else None ) # Auto-refresh state. Optional: only crawls that re-scrape themselves # carry the block. Built from the whole payload rather than the nested # dict because CrawlerRefreshState accepts either envelope. self.refresh: Optional[CrawlerRefreshState] = ( CrawlerRefreshState(response_data) if isinstance(response_data.get('refresh'), dict) else None ) @property def is_complete(self) -> bool: """Whether the crawler reached DONE with is_success=True.""" return self.status == self.STATUS_DONE and self.is_success is True @property def is_running(self) -> bool: """Whether the crawler is currently PENDING or RUNNING.""" return self.status in (self.STATUS_PENDING, self.STATUS_RUNNING) @property def is_failed(self) -> bool: """Whether the crawler reached DONE with is_success=False.""" return self.status == self.STATUS_DONE and self.is_success is False @property def is_cancelled(self) -> bool: """Whether the crawler was cancelled.""" return self.status == self.STATUS_CANCELLED @property def progress_pct(self) -> float: """ Visited/extracted ratio as a percentage (0-100). Returns 0.0 when no URLs have been extracted yet. """ if self.state.urls_extracted == 0: return 0.0 return (self.state.urls_visited / self.state.urls_extracted) * 100 def __repr__(self): return (f"CrawlerStatusResponse(uuid={self.uuid}, status={self.status}, " f"progress={self.progress_pct:.1f}%, " f"visited={self.state.urls_visited}/{self.state.urls_extracted})")Response from checking crawler job status.
Returned by :py:meth:
ScrapflyClient.get_crawl_status(). Provides real-time progress tracking for crawler jobs.Field names match the wire format. Scrapfly is the source of truth; the Go and TypeScript SDKs expose identical names. Access state counters via the nested
stateattribute:>>> status.state.urls_visited 12 >>> status.state.urls_extracted 34Attributes
uuid- Crawler job UUID.
status- Current status (
PENDING,RUNNING,DONE,CANCELLED). is_success- Whether the crawler job completed successfully (
Nonewhile running). is_finished- Whether the crawler job has finished (regardless of success/failure).
state- :class:
CrawlerState— all the per-crawl counters and timings. search- :class:
CrawlerSearchStatewhen the crawl was started withsearch=True, otherwiseNone. refresh- :class:
CrawlerRefreshStatewhen the crawl re-scrapes itself on a period, otherwiseNone.
Initialize from API response.
Strict parsing: required fields (
crawler_uuid,status,is_success,is_finished, and the documentedstate.*metrics) are read with direct access so missing keys raiseKeyErrorat parse time. This catches API contract drift loud and early.Args
response_data- Raw API response dictionary.
Class variables
var STATUS_CANCELLEDvar STATUS_DONEvar STATUS_PENDINGvar STATUS_RUNNING
Instance variables
prop is_cancelled : bool-
Expand source code
@property def is_cancelled(self) -> bool: """Whether the crawler was cancelled.""" return self.status == self.STATUS_CANCELLEDWhether the crawler was cancelled.
prop is_complete : bool-
Expand source code
@property def is_complete(self) -> bool: """Whether the crawler reached DONE with is_success=True.""" return self.status == self.STATUS_DONE and self.is_success is TrueWhether the crawler reached DONE with is_success=True.
prop is_failed : bool-
Expand source code
@property def is_failed(self) -> bool: """Whether the crawler reached DONE with is_success=False.""" return self.status == self.STATUS_DONE and self.is_success is FalseWhether the crawler reached DONE with is_success=False.
prop is_running : bool-
Expand source code
@property def is_running(self) -> bool: """Whether the crawler is currently PENDING or RUNNING.""" return self.status in (self.STATUS_PENDING, self.STATUS_RUNNING)Whether the crawler is currently PENDING or RUNNING.
prop progress_pct : float-
Expand source code
@property def progress_pct(self) -> float: """ Visited/extracted ratio as a percentage (0-100). Returns 0.0 when no URLs have been extracted yet. """ if self.state.urls_extracted == 0: return 0.0 return (self.state.urls_visited / self.state.urls_extracted) * 100Visited/extracted ratio as a percentage (0-100).
Returns 0.0 when no URLs have been extracted yet.
class CrawlerUpdatedDocuments (updated: List[str] = <factory>,
removed: List[str] = <factory>,
truncated: bool = False)-
Expand source code
@dataclass class CrawlerUpdatedDocuments: """ The URLs one refresh run changed. Both lists are capped by Scrapfly at 100 URLs, so a run that changed more than that arrives with ``truncated`` set and the counts on :class:`CrawlerRefreshEntry` describing the whole run. There is no cursor: the event is a notification, the crawl itself is the export. Attributes: updated: Re-indexed URLs, added and changed alike. Which of the two a URL was only survives in the counts. removed: URLs dropped from the crawl because they are gone. truncated: Whether either list was cut at the cap. """ updated: List[str] = field(default_factory=list) removed: List[str] = field(default_factory=list) truncated: bool = False @classmethod def from_dict(cls, data: Dict[str, Any]) -> 'CrawlerUpdatedDocuments': return cls( updated=list(data.get('updated') or []), removed=list(data.get('removed') or []), truncated=bool(data.get('truncated')), )The URLs one refresh run changed.
Both lists are capped by Scrapfly at 100 URLs, so a run that changed more than that arrives with
truncatedset and the counts on :class:CrawlerRefreshEntrydescribing the whole run. There is no cursor: the event is a notification, the crawl itself is the export.Attributes
updated- Re-indexed URLs, added and changed alike. Which of the two a URL was only survives in the counts.
removed- URLs dropped from the crawl because they are gone.
truncated- Whether either list was cut at the cap.
Static methods
def from_dict(data: Dict[str, Any]) ‑> CrawlerUpdatedDocuments
Instance variables
var removed : List[str]var truncated : boolvar updated : List[str]
class CrawlerUpdatedWebhook (event: str,
crawler_uuid: str,
project: str,
env: str,
action: str,
state: CrawlerState,
seed_url: str,
status_link: str,
refresh: CrawlerRefreshEntry,
documents: CrawlerUpdatedDocuments)-
Expand source code
@dataclass class CrawlerUpdatedWebhook(CrawlerWebhookBase): """ Payload for the ``crawler_updated`` event. Emitted once per auto-refresh run that changed at least one page. A run over a site that stood still, and a run that failed outright, change nothing and are not delivered, so receiving this event is by itself proof of a diff. Attributes: seed_url: The root URL the crawl was started from. status_link: URL to fetch the live crawler status. refresh: The run, as the same row the refresh timeline keeps. ``sample_updated`` / ``sample_removed`` are empty on this block: the webhook carries the URLs in ``documents`` instead, at a higher cap. documents: The changed URLs, capped. """ seed_url: str status_link: str refresh: CrawlerRefreshEntry documents: CrawlerUpdatedDocuments @classmethod def from_payload(cls, event: str, payload: Dict[str, Any]) -> 'CrawlerUpdatedWebhook': base = cls._parse_base(event, payload) return cls( **base, seed_url=payload['seed_url'], status_link=payload['links']['status'], refresh=CrawlerRefreshEntry.from_dict(payload['refresh']), documents=CrawlerUpdatedDocuments.from_dict(payload['documents']), )Payload for the
crawler_updatedevent.Emitted once per auto-refresh run that changed at least one page. A run over a site that stood still, and a run that failed outright, change nothing and are not delivered, so receiving this event is by itself proof of a diff.
Attributes
seed_url- The root URL the crawl was started from.
status_link- URL to fetch the live crawler status.
refresh- The run, as the same row the refresh timeline keeps.
sample_updated/sample_removedare empty on this block: the webhook carries the URLs indocumentsinstead, at a higher cap. documents- The changed URLs, capped.
Ancestors
Static methods
def from_payload(event: str, payload: Dict[str, Any]) ‑> CrawlerUpdatedWebhook
Instance variables
var documents : CrawlerUpdatedDocumentsvar refresh : CrawlerRefreshEntryvar seed_url : strvar status_link : str
class CrawlerUrlDiscoveredWebhook (event: str,
crawler_uuid: str,
project: str,
env: str,
action: str,
state: CrawlerState,
origin: str,
discovered_urls: List[str])-
Expand source code
@dataclass class CrawlerUrlDiscoveredWebhook(CrawlerWebhookBase): """ Payload for the ``crawler_url_discovered`` event. Emitted when the crawler extracts one or more new URLs from a source. Attributes: origin: How the URLs were discovered (e.g. ``"navigation"``, ``"sitemap"``). discovered_urls: The newly-discovered URLs as a list. """ origin: str discovered_urls: List[str] @classmethod def from_payload(cls, event: str, payload: Dict[str, Any]) -> 'CrawlerUrlDiscoveredWebhook': base = cls._parse_base(event, payload) return cls( **base, origin=payload['origin'], discovered_urls=payload['discovered_urls'], )Payload for the
crawler_url_discoveredevent.Emitted when the crawler extracts one or more new URLs from a source.
Attributes
origin- How the URLs were discovered (e.g.
"navigation","sitemap"). discovered_urls- The newly-discovered URLs as a list.
Ancestors
Static methods
def from_payload(event: str, payload: Dict[str, Any]) ‑> CrawlerUrlDiscoveredWebhook
Instance variables
var discovered_urls : List[str]var origin : str
class CrawlerUrlEntry (url: str, status: str, reason: str | None = None)-
Expand source code
class CrawlerUrlEntry: """ Single URL entry from ``GET /crawl/{uuid}/urls``. The endpoint streams one record per line as ``text/plain``. For ``visited`` and ``pending`` URLs each line is just the URL; for ``failed`` or ``skipped`` URLs the line is ``url,reason``. Streaming text is used because this endpoint is expected to scale to millions of records per job — JSON is not a suitable wire format at that volume. Attributes: url: The crawled URL status: The filter status used by the caller (``visited``, ``pending``, ``failed`` or ``skipped``). Echoed from the request parameter so downstream code can disambiguate mixed buffers. reason: Only set for ``failed`` / ``skipped`` URLs; ``None`` otherwise. """ __slots__ = ('url', 'status', 'reason') def __init__(self, url: str, status: str, reason: Optional[str] = None): assert isinstance(url, str) and url, ( f"CrawlerUrlEntry: url must be a non-empty string, got {url!r}" ) assert isinstance(status, str) and status, ( f"CrawlerUrlEntry: status must be a non-empty string, got {status!r}" ) self.url = url self.status = status self.reason = reason def __repr__(self): if self.reason is not None: return f"CrawlerUrlEntry(url={self.url!r}, status={self.status!r}, reason={self.reason!r})" return f"CrawlerUrlEntry(url={self.url!r}, status={self.status!r})"Single URL entry from
GET /crawl/{uuid}/urls.The endpoint streams one record per line as
text/plain. ForvisitedandpendingURLs each line is just the URL; forfailedorskippedURLs the line isurl,reason. Streaming text is used because this endpoint is expected to scale to millions of records per job — JSON is not a suitable wire format at that volume.Attributes
url- The crawled URL
status- The filter status used by the caller (
visited,pending,failedorskipped). Echoed from the request parameter so downstream code can disambiguate mixed buffers. reason- Only set for
failed/skippedURLs;Noneotherwise.
Instance variables
var reason-
Expand source code
class CrawlerUrlEntry: """ Single URL entry from ``GET /crawl/{uuid}/urls``. The endpoint streams one record per line as ``text/plain``. For ``visited`` and ``pending`` URLs each line is just the URL; for ``failed`` or ``skipped`` URLs the line is ``url,reason``. Streaming text is used because this endpoint is expected to scale to millions of records per job — JSON is not a suitable wire format at that volume. Attributes: url: The crawled URL status: The filter status used by the caller (``visited``, ``pending``, ``failed`` or ``skipped``). Echoed from the request parameter so downstream code can disambiguate mixed buffers. reason: Only set for ``failed`` / ``skipped`` URLs; ``None`` otherwise. """ __slots__ = ('url', 'status', 'reason') def __init__(self, url: str, status: str, reason: Optional[str] = None): assert isinstance(url, str) and url, ( f"CrawlerUrlEntry: url must be a non-empty string, got {url!r}" ) assert isinstance(status, str) and status, ( f"CrawlerUrlEntry: status must be a non-empty string, got {status!r}" ) self.url = url self.status = status self.reason = reason def __repr__(self): if self.reason is not None: return f"CrawlerUrlEntry(url={self.url!r}, status={self.status!r}, reason={self.reason!r})" return f"CrawlerUrlEntry(url={self.url!r}, status={self.status!r})" var status-
Expand source code
class CrawlerUrlEntry: """ Single URL entry from ``GET /crawl/{uuid}/urls``. The endpoint streams one record per line as ``text/plain``. For ``visited`` and ``pending`` URLs each line is just the URL; for ``failed`` or ``skipped`` URLs the line is ``url,reason``. Streaming text is used because this endpoint is expected to scale to millions of records per job — JSON is not a suitable wire format at that volume. Attributes: url: The crawled URL status: The filter status used by the caller (``visited``, ``pending``, ``failed`` or ``skipped``). Echoed from the request parameter so downstream code can disambiguate mixed buffers. reason: Only set for ``failed`` / ``skipped`` URLs; ``None`` otherwise. """ __slots__ = ('url', 'status', 'reason') def __init__(self, url: str, status: str, reason: Optional[str] = None): assert isinstance(url, str) and url, ( f"CrawlerUrlEntry: url must be a non-empty string, got {url!r}" ) assert isinstance(status, str) and status, ( f"CrawlerUrlEntry: status must be a non-empty string, got {status!r}" ) self.url = url self.status = status self.reason = reason def __repr__(self): if self.reason is not None: return f"CrawlerUrlEntry(url={self.url!r}, status={self.status!r}, reason={self.reason!r})" return f"CrawlerUrlEntry(url={self.url!r}, status={self.status!r})" var url-
Expand source code
class CrawlerUrlEntry: """ Single URL entry from ``GET /crawl/{uuid}/urls``. The endpoint streams one record per line as ``text/plain``. For ``visited`` and ``pending`` URLs each line is just the URL; for ``failed`` or ``skipped`` URLs the line is ``url,reason``. Streaming text is used because this endpoint is expected to scale to millions of records per job — JSON is not a suitable wire format at that volume. Attributes: url: The crawled URL status: The filter status used by the caller (``visited``, ``pending``, ``failed`` or ``skipped``). Echoed from the request parameter so downstream code can disambiguate mixed buffers. reason: Only set for ``failed`` / ``skipped`` URLs; ``None`` otherwise. """ __slots__ = ('url', 'status', 'reason') def __init__(self, url: str, status: str, reason: Optional[str] = None): assert isinstance(url, str) and url, ( f"CrawlerUrlEntry: url must be a non-empty string, got {url!r}" ) assert isinstance(status, str) and status, ( f"CrawlerUrlEntry: status must be a non-empty string, got {status!r}" ) self.url = url self.status = status self.reason = reason def __repr__(self): if self.reason is not None: return f"CrawlerUrlEntry(url={self.url!r}, status={self.status!r}, reason={self.reason!r})" return f"CrawlerUrlEntry(url={self.url!r}, status={self.status!r})"
class CrawlerUrlFailedWebhook (event: str,
crawler_uuid: str,
project: str,
env: str,
action: str,
state: CrawlerState,
url: str,
error: str,
scrape_config: Dict[str, Any],
log_link: str | None,
scrape_link: str)-
Expand source code
@dataclass class CrawlerUrlFailedWebhook(CrawlerWebhookBase): """ Payload for the ``crawler_url_failed`` event. Emitted when a URL cannot be crawled (network error, scrape error, blocked, etc.). Attributes: url: The URL that failed. error: The scrapfly error code (e.g. ``ERR::SCRAPE::NETWORK_ERROR``). scrape_config: The scrape config that was used for the failed attempt. log_link: URL to the full scrape log for this failure. Can be ``None`` — Scrapfly emits ``null`` when no log was recorded (e.g. the failure happened before the request was ever executed). scrape_link: URL that re-runs the same scrape as a one-off. Always present on the wire (non-nullable). """ url: str error: str scrape_config: Dict[str, Any] log_link: Optional[str] scrape_link: str @classmethod def from_payload(cls, event: str, payload: Dict[str, Any]) -> 'CrawlerUrlFailedWebhook': base = cls._parse_base(event, payload) return cls( **base, url=payload['url'], error=payload['error'], scrape_config=payload['scrape_config'], log_link=payload['links'].get('log'), scrape_link=payload['links']['scrape'], )Payload for the
crawler_url_failedevent.Emitted when a URL cannot be crawled (network error, scrape error, blocked, etc.).
Attributes
url- The URL that failed.
error- The scrapfly error code (e.g.
ERR::SCRAPE::NETWORK_ERROR). scrape_config- The scrape config that was used for the failed attempt.
log_link- URL to the full scrape log for this failure. Can be
None— Scrapfly emitsnullwhen no log was recorded (e.g. the failure happened before the request was ever executed). scrape_link- URL that re-runs the same scrape as a one-off. Always present on the wire (non-nullable).
Ancestors
Static methods
def from_payload(event: str, payload: Dict[str, Any]) ‑> CrawlerUrlFailedWebhook
Instance variables
var error : strvar log_link : str | Nonevar scrape_config : Dict[str, Any]var scrape_link : strvar url : str
class CrawlerUrlSkippedWebhook (event: str,
crawler_uuid: str,
project: str,
env: str,
action: str,
state: CrawlerState,
urls: Dict[str, str])-
Expand source code
@dataclass class CrawlerUrlSkippedWebhook(CrawlerWebhookBase): """ Payload for the ``crawler_url_skipped`` event. Emitted in a single batch when the crawler decides to skip a set of URLs (e.g. when reaching ``page_limit`` with discovered-but-unvisited URLs still in the queue). Attributes: urls: Mapping from URL to the reason it was skipped (e.g. ``"page_limit"``, ``"excluded"``, ``"robots_txt"``). """ urls: Dict[str, str] @classmethod def from_payload(cls, event: str, payload: Dict[str, Any]) -> 'CrawlerUrlSkippedWebhook': base = cls._parse_base(event, payload) return cls(**base, urls=payload['urls'])Payload for the
crawler_url_skippedevent.Emitted in a single batch when the crawler decides to skip a set of URLs (e.g. when reaching
page_limitwith discovered-but-unvisited URLs still in the queue).Attributes
urls- Mapping from URL to the reason it was skipped
(e.g.
"page_limit","excluded","robots_txt").
Ancestors
Static methods
def from_payload(event: str, payload: Dict[str, Any]) ‑> CrawlerUrlSkippedWebhook
Instance variables
var urls : Dict[str, str]
class CrawlerUrlVisitedWebhook (event: str,
crawler_uuid: str,
project: str,
env: str,
action: str,
state: CrawlerState,
url: str,
scrape: CrawlerScrapeResult)-
Expand source code
@dataclass class CrawlerUrlVisitedWebhook(CrawlerWebhookBase): """ Payload for the ``crawler_url_visited`` event. Emitted after each URL has been successfully scraped. Attributes: url: The URL that was just visited. scrape: Scrape result details (status code, country, log link, content). """ url: str scrape: CrawlerScrapeResult @classmethod def from_payload(cls, event: str, payload: Dict[str, Any]) -> 'CrawlerUrlVisitedWebhook': base = cls._parse_base(event, payload) return cls( **base, url=payload['url'], scrape=CrawlerScrapeResult.from_dict(payload['scrape']), )Payload for the
crawler_url_visitedevent.Emitted after each URL has been successfully scraped.
Attributes
url- The URL that was just visited.
scrape- Scrape result details (status code, country, log link, content).
Ancestors
Static methods
def from_payload(event: str, payload: Dict[str, Any]) ‑> CrawlerUrlVisitedWebhook
Instance variables
var scrape : CrawlerScrapeResultvar url : str
class CrawlerUrlsResponse (urls: List[ForwardRef('CrawlerUrlEntry')],
page: int,
per_page: int)-
Expand source code
class CrawlerUrlsResponse: """ Response from ``GET /crawl/{crawler_uuid}/urls``. The server returns a streaming ``text/plain`` body with one record per line. This class parses that stream into a materialised ``List`` of :class:`CrawlerUrlEntry` records for caller convenience. Pagination: the wire protocol carries no global ``total``, and the API forwards only the status filter to the crawler, so ``page`` and ``per_page`` are echoes of the caller's request parameters over a body that already holds the whole server-side page. Attributes: urls: List of :class:`CrawlerUrlEntry` records on this page page: 1-based page number (echoed from the request) per_page: Page size (echoed from the request) """ __slots__ = ('urls', 'page', 'per_page') def __init__(self, urls: List['CrawlerUrlEntry'], page: int, per_page: int): self.urls = urls self.page = page self.per_page = per_page @classmethod def from_text( cls, body: str, status_hint: str, page: int, per_page: int, ) -> 'CrawlerUrlsResponse': """ Parse the raw text body returned by ``GET /crawl/{uuid}/urls``. - Empty lines are ignored (trailing newlines, blank records). - For ``visited`` / ``pending`` status each line is one URL. - For ``failed`` / ``skipped`` status each line is ``url,reason``. - When the caller passed no ``status`` filter, the server defaults to ``visited``; the caller is expected to pass that as ``status_hint`` so every parsed record gets the right status tag. Args: body: Raw response body text. status_hint: The status filter the caller used. page: Caller-provided page (echoed on the response object). per_page: Caller-provided per_page (echoed on the response object). """ entries: List[CrawlerUrlEntry] = [] for raw_line in body.splitlines(): line = raw_line.strip() if not line: continue if status_hint in ('visited', 'pending'): entries.append(CrawlerUrlEntry(url=line, status=status_hint)) else: # `url,reason` — split on the first comma only. URLs never # contain an unencoded comma in the path/query, so this is # unambiguous. comma_idx = line.find(',') if comma_idx == -1: entries.append(CrawlerUrlEntry(url=line, status=status_hint)) else: entries.append( CrawlerUrlEntry( url=line[:comma_idx], status=status_hint, reason=line[comma_idx + 1:] or None, ) ) return cls(entries, page, per_page) def __len__(self) -> int: return len(self.urls) def __iter__(self) -> Iterator[CrawlerUrlEntry]: return iter(self.urls) def __repr__(self): return ( f"CrawlerUrlsResponse(page={self.page}, per_page={self.per_page}, " f"urls={len(self.urls)})" )Response from
GET /crawl/{crawler_uuid}/urls.The server returns a streaming
text/plainbody with one record per line. This class parses that stream into a materialisedListof :class:CrawlerUrlEntryrecords for caller convenience.Pagination: the wire protocol carries no global
total, and the API forwards only the status filter to the crawler, sopageandper_pageare echoes of the caller's request parameters over a body that already holds the whole server-side page.Attributes
urls- List of :class:
CrawlerUrlEntryrecords on this page page- 1-based page number (echoed from the request)
per_page- Page size (echoed from the request)
Static methods
def from_text(body: str, status_hint: str, page: int, per_page: int) ‑> CrawlerUrlsResponse-
Parse the raw text body returned by
GET /crawl/{uuid}/urls.- Empty lines are ignored (trailing newlines, blank records).
- For
visited/pendingstatus each line is one URL. - For
failed/skippedstatus each line isurl,reason. - When the caller passed no
statusfilter, the server defaults tovisited; the caller is expected to pass that asstatus_hintso every parsed record gets the right status tag.
Args
body- Raw response body text.
status_hint- The status filter the caller used.
page- Caller-provided page (echoed on the response object).
per_page- Caller-provided per_page (echoed on the response object).
Instance variables
var page-
Expand source code
class CrawlerUrlsResponse: """ Response from ``GET /crawl/{crawler_uuid}/urls``. The server returns a streaming ``text/plain`` body with one record per line. This class parses that stream into a materialised ``List`` of :class:`CrawlerUrlEntry` records for caller convenience. Pagination: the wire protocol carries no global ``total``, and the API forwards only the status filter to the crawler, so ``page`` and ``per_page`` are echoes of the caller's request parameters over a body that already holds the whole server-side page. Attributes: urls: List of :class:`CrawlerUrlEntry` records on this page page: 1-based page number (echoed from the request) per_page: Page size (echoed from the request) """ __slots__ = ('urls', 'page', 'per_page') def __init__(self, urls: List['CrawlerUrlEntry'], page: int, per_page: int): self.urls = urls self.page = page self.per_page = per_page @classmethod def from_text( cls, body: str, status_hint: str, page: int, per_page: int, ) -> 'CrawlerUrlsResponse': """ Parse the raw text body returned by ``GET /crawl/{uuid}/urls``. - Empty lines are ignored (trailing newlines, blank records). - For ``visited`` / ``pending`` status each line is one URL. - For ``failed`` / ``skipped`` status each line is ``url,reason``. - When the caller passed no ``status`` filter, the server defaults to ``visited``; the caller is expected to pass that as ``status_hint`` so every parsed record gets the right status tag. Args: body: Raw response body text. status_hint: The status filter the caller used. page: Caller-provided page (echoed on the response object). per_page: Caller-provided per_page (echoed on the response object). """ entries: List[CrawlerUrlEntry] = [] for raw_line in body.splitlines(): line = raw_line.strip() if not line: continue if status_hint in ('visited', 'pending'): entries.append(CrawlerUrlEntry(url=line, status=status_hint)) else: # `url,reason` — split on the first comma only. URLs never # contain an unencoded comma in the path/query, so this is # unambiguous. comma_idx = line.find(',') if comma_idx == -1: entries.append(CrawlerUrlEntry(url=line, status=status_hint)) else: entries.append( CrawlerUrlEntry( url=line[:comma_idx], status=status_hint, reason=line[comma_idx + 1:] or None, ) ) return cls(entries, page, per_page) def __len__(self) -> int: return len(self.urls) def __iter__(self) -> Iterator[CrawlerUrlEntry]: return iter(self.urls) def __repr__(self): return ( f"CrawlerUrlsResponse(page={self.page}, per_page={self.per_page}, " f"urls={len(self.urls)})" ) var per_page-
Expand source code
class CrawlerUrlsResponse: """ Response from ``GET /crawl/{crawler_uuid}/urls``. The server returns a streaming ``text/plain`` body with one record per line. This class parses that stream into a materialised ``List`` of :class:`CrawlerUrlEntry` records for caller convenience. Pagination: the wire protocol carries no global ``total``, and the API forwards only the status filter to the crawler, so ``page`` and ``per_page`` are echoes of the caller's request parameters over a body that already holds the whole server-side page. Attributes: urls: List of :class:`CrawlerUrlEntry` records on this page page: 1-based page number (echoed from the request) per_page: Page size (echoed from the request) """ __slots__ = ('urls', 'page', 'per_page') def __init__(self, urls: List['CrawlerUrlEntry'], page: int, per_page: int): self.urls = urls self.page = page self.per_page = per_page @classmethod def from_text( cls, body: str, status_hint: str, page: int, per_page: int, ) -> 'CrawlerUrlsResponse': """ Parse the raw text body returned by ``GET /crawl/{uuid}/urls``. - Empty lines are ignored (trailing newlines, blank records). - For ``visited`` / ``pending`` status each line is one URL. - For ``failed`` / ``skipped`` status each line is ``url,reason``. - When the caller passed no ``status`` filter, the server defaults to ``visited``; the caller is expected to pass that as ``status_hint`` so every parsed record gets the right status tag. Args: body: Raw response body text. status_hint: The status filter the caller used. page: Caller-provided page (echoed on the response object). per_page: Caller-provided per_page (echoed on the response object). """ entries: List[CrawlerUrlEntry] = [] for raw_line in body.splitlines(): line = raw_line.strip() if not line: continue if status_hint in ('visited', 'pending'): entries.append(CrawlerUrlEntry(url=line, status=status_hint)) else: # `url,reason` — split on the first comma only. URLs never # contain an unencoded comma in the path/query, so this is # unambiguous. comma_idx = line.find(',') if comma_idx == -1: entries.append(CrawlerUrlEntry(url=line, status=status_hint)) else: entries.append( CrawlerUrlEntry( url=line[:comma_idx], status=status_hint, reason=line[comma_idx + 1:] or None, ) ) return cls(entries, page, per_page) def __len__(self) -> int: return len(self.urls) def __iter__(self) -> Iterator[CrawlerUrlEntry]: return iter(self.urls) def __repr__(self): return ( f"CrawlerUrlsResponse(page={self.page}, per_page={self.per_page}, " f"urls={len(self.urls)})" ) var urls-
Expand source code
class CrawlerUrlsResponse: """ Response from ``GET /crawl/{crawler_uuid}/urls``. The server returns a streaming ``text/plain`` body with one record per line. This class parses that stream into a materialised ``List`` of :class:`CrawlerUrlEntry` records for caller convenience. Pagination: the wire protocol carries no global ``total``, and the API forwards only the status filter to the crawler, so ``page`` and ``per_page`` are echoes of the caller's request parameters over a body that already holds the whole server-side page. Attributes: urls: List of :class:`CrawlerUrlEntry` records on this page page: 1-based page number (echoed from the request) per_page: Page size (echoed from the request) """ __slots__ = ('urls', 'page', 'per_page') def __init__(self, urls: List['CrawlerUrlEntry'], page: int, per_page: int): self.urls = urls self.page = page self.per_page = per_page @classmethod def from_text( cls, body: str, status_hint: str, page: int, per_page: int, ) -> 'CrawlerUrlsResponse': """ Parse the raw text body returned by ``GET /crawl/{uuid}/urls``. - Empty lines are ignored (trailing newlines, blank records). - For ``visited`` / ``pending`` status each line is one URL. - For ``failed`` / ``skipped`` status each line is ``url,reason``. - When the caller passed no ``status`` filter, the server defaults to ``visited``; the caller is expected to pass that as ``status_hint`` so every parsed record gets the right status tag. Args: body: Raw response body text. status_hint: The status filter the caller used. page: Caller-provided page (echoed on the response object). per_page: Caller-provided per_page (echoed on the response object). """ entries: List[CrawlerUrlEntry] = [] for raw_line in body.splitlines(): line = raw_line.strip() if not line: continue if status_hint in ('visited', 'pending'): entries.append(CrawlerUrlEntry(url=line, status=status_hint)) else: # `url,reason` — split on the first comma only. URLs never # contain an unencoded comma in the path/query, so this is # unambiguous. comma_idx = line.find(',') if comma_idx == -1: entries.append(CrawlerUrlEntry(url=line, status=status_hint)) else: entries.append( CrawlerUrlEntry( url=line[:comma_idx], status=status_hint, reason=line[comma_idx + 1:] or None, ) ) return cls(entries, page, per_page) def __len__(self) -> int: return len(self.urls) def __iter__(self) -> Iterator[CrawlerUrlEntry]: return iter(self.urls) def __repr__(self): return ( f"CrawlerUrlsResponse(page={self.page}, per_page={self.per_page}, " f"urls={len(self.urls)})" )
class CrawlerWebhookBase (event: str,
crawler_uuid: str,
project: str,
env: str,
action: str,
state: CrawlerState)-
Expand source code
@dataclass class CrawlerWebhookBase: """ Common fields carried by every crawler webhook payload. Attributes: event: The wire event name (``crawler_started``, etc.). crawler_uuid: The crawler job UUID. project: Project slug the crawler belongs to. env: Environment (``LIVE`` or ``TEST``). action: Short action tag emitted by Scrapfly (``started``, ``visited``, ``skipped``, ``url_discovery``, ``failed``, ``stopped``, ``cancelled``, ``finished``). state: Nested state counters at the moment the webhook was emitted. """ event: str crawler_uuid: str project: str env: str action: str state: CrawlerState @staticmethod def _parse_base(event: str, payload: Dict[str, Any]) -> Dict[str, Any]: """ Extract the 5 fields every webhook carries. Used by subclass ``from_payload()`` factories. """ return { 'event': event, 'crawler_uuid': payload['crawler_uuid'], 'project': payload['project'], 'env': payload['env'], 'action': payload['action'], 'state': CrawlerState(payload['state']), }Common fields carried by every crawler webhook payload.
Attributes
event- The wire event name (
crawler_started, etc.). crawler_uuid- The crawler job UUID.
project- Project slug the crawler belongs to.
env- Environment (
LIVEorTEST). action- Short action tag emitted by Scrapfly
(
started,visited,skipped,url_discovery,failed,stopped,cancelled,finished). state- Nested state counters at the moment the webhook was emitted.
Subclasses
- CrawlerLifecycleWebhook
- CrawlerSearchWebhook
- CrawlerUpdatedWebhook
- CrawlerUrlDiscoveredWebhook
- CrawlerUrlFailedWebhook
- CrawlerUrlSkippedWebhook
- CrawlerUrlVisitedWebhook
Instance variables
var action : strvar crawler_uuid : strvar env : strvar event : strvar project : strvar state : CrawlerState
class CrawlerWebhookEvent (value, names=None, *, module=None, qualname=None, type=None, start=1)-
Expand source code
class CrawlerWebhookEvent(str, Enum): """ Crawler webhook event names. These MUST stay in sync with class ``WebhookEvents``. Scrapfly is the source of truth. """ CRAWLER_STARTED = 'crawler_started' CRAWLER_STOPPED = 'crawler_stopped' CRAWLER_CANCELLED = 'crawler_cancelled' CRAWLER_FINISHED = 'crawler_finished' CRAWLER_URL_VISITED = 'crawler_url_visited' CRAWLER_URL_SKIPPED = 'crawler_url_skipped' CRAWLER_URL_DISCOVERED = 'crawler_url_discovered' CRAWLER_URL_FAILED = 'crawler_url_failed' CRAWLER_SEARCH_READY = 'crawler_search_ready' CRAWLER_SEARCH_FAILED = 'crawler_search_failed' CRAWLER_UPDATED = 'crawler_updated'Crawler webhook event names.
These MUST stay in sync with class
WebhookEvents. Scrapfly is the source of truth.Ancestors
- builtins.str
- enum.Enum
Class variables
var CRAWLER_CANCELLEDvar CRAWLER_FINISHEDvar CRAWLER_SEARCH_FAILEDvar CRAWLER_SEARCH_READYvar CRAWLER_STARTEDvar CRAWLER_STOPPEDvar CRAWLER_UPDATEDvar CRAWLER_URL_DISCOVEREDvar CRAWLER_URL_FAILEDvar CRAWLER_URL_SKIPPEDvar CRAWLER_URL_VISITED
class CreateScheduleRequest (webhook_name: str = '',
recurrence: ScheduleRecurrence | None = None,
scheduled_date: str | None = None,
allow_concurrency: bool = False,
retry_on_failure: bool = False,
max_retries: int | None = None,
notes: str | None = None)-
Expand source code
@dataclass class CreateScheduleRequest: """Public-facing request envelope for creating a schedule. The kind-specific configuration (scrape_config / screenshot_config / crawler_config) is supplied as a separate argument by the matching ``create_*_schedule`` method. """ webhook_name: str = "" recurrence: Optional[ScheduleRecurrence] = None scheduled_date: Optional[str] = None allow_concurrency: bool = False retry_on_failure: bool = False max_retries: Optional[int] = None notes: Optional[str] = NonePublic-facing request envelope for creating a schedule.
The kind-specific configuration (scrape_config / screenshot_config / crawler_config) is supplied as a separate argument by the matching
create_*_schedulemethod.Instance variables
var allow_concurrency : boolvar max_retries : int | Nonevar notes : str | Nonevar recurrence : ScheduleRecurrence | Nonevar retry_on_failure : boolvar scheduled_date : str | Nonevar webhook_name : str
class EncoderError (content: str)-
Expand source code
class EncoderError(BaseException): def __init__(self, content:str): self.content = content super().__init__() def __str__(self) -> str: return self.content def __repr__(self): return "Invalid payload: %s" % self.contentCommon base class for all exceptions
Ancestors
- builtins.BaseException
class ErrorFactory-
Expand source code
class ErrorFactory: RESOURCE_TO_ERROR = { ScrapflyError.RESOURCE_SCRAPE: ScrapflyScrapeError, ScrapflyError.RESOURCE_WEBHOOK: ScrapflyWebhookError, ScrapflyError.RESOURCE_PROXY: ScrapflyProxyError, ScrapflyError.RESOURCE_SCHEDULE: ScrapflyScheduleError, ScrapflyError.RESOURCE_ASP: ScrapflyAspError, ScrapflyError.RESOURCE_SESSION: ScrapflySessionError } # Notable http error has own class for more convenience # Only applicable for generic API error HTTP_STATUS_TO_ERROR = { 401: BadApiKeyError, 402: PaymentRequired, 429: TooManyRequest } @staticmethod def _get_resource(code: str) -> Optional[str]: # Codes are ERR::<RESOURCE>::<REASON>, but the segment count is the # API's to change, so index rather than unpack. if isinstance(code, str) and '::' in code: return code.split('::')[1] return None @staticmethod def create(api_response: 'ScrapeApiResponse'): is_retryable = False kind = ScrapflyError.KIND_HTTP_BAD_RESPONSE if api_response.success is False else ScrapflyError.KIND_SCRAPFLY_ERROR http_code = api_response.status_code retry_delay = 5 retry_times = 3 description = None error_url = 'https://scrapfly.io/docs/scrape-api/errors#api' code = api_response.error['code'] if code == 'ERR::SCRAPE::BAD_UPSTREAM_RESPONSE': http_code = api_response.scrape_result['status_code'] if 'description' in api_response.error: description = api_response.error['description'] message = '%s %s %s' % (str(http_code), code, api_response.error['message']) if 'doc_url' in api_response.error: error_url = api_response.error['doc_url'] if 'retryable' in api_response.error: is_retryable = api_response.error['retryable'] resource = ErrorFactory._get_resource(code=code) if is_retryable is True: if 'X-Retry' in api_response.headers: retry_delay = int(api_response.headers['Retry-After']) message = '%s: %s' % (message, description) if description else message if retry_delay is not None and is_retryable is True: message = '%s. Retry delay : %s seconds' % (message, str(retry_delay)) args = { 'message': message, 'code': code, 'http_status_code': http_code, 'is_retryable': is_retryable, 'api_response': api_response, 'resource': resource, 'retry_delay': retry_delay, 'retry_times': retry_times, 'documentation_url': error_url, 'request': api_response.request, 'response': api_response.response } if kind == ScrapflyError.KIND_HTTP_BAD_RESPONSE: if http_code >= 500: return ApiHttpServerError(**args) is_scraper_api_error = resource in ErrorFactory.RESOURCE_TO_ERROR if http_code in ErrorFactory.HTTP_STATUS_TO_ERROR and not is_scraper_api_error: return ErrorFactory.HTTP_STATUS_TO_ERROR[http_code](**args) if is_scraper_api_error: return ErrorFactory.RESOURCE_TO_ERROR[resource](**args) return ApiHttpClientError(**args) elif kind == ScrapflyError.KIND_SCRAPFLY_ERROR: if code == 'ERR::SCRAPE::BAD_UPSTREAM_RESPONSE': if http_code >= 500: return UpstreamHttpServerError(**args) if http_code >= 400: return UpstreamHttpClientError(**args) if resource in ErrorFactory.RESOURCE_TO_ERROR: return ErrorFactory.RESOURCE_TO_ERROR[resource](**args) return ScrapflyError(**args)Class variables
var HTTP_STATUS_TO_ERRORvar RESOURCE_TO_ERROR
Static methods
def create(api_response: ScrapeApiResponse)-
Expand source code
@staticmethod def create(api_response: 'ScrapeApiResponse'): is_retryable = False kind = ScrapflyError.KIND_HTTP_BAD_RESPONSE if api_response.success is False else ScrapflyError.KIND_SCRAPFLY_ERROR http_code = api_response.status_code retry_delay = 5 retry_times = 3 description = None error_url = 'https://scrapfly.io/docs/scrape-api/errors#api' code = api_response.error['code'] if code == 'ERR::SCRAPE::BAD_UPSTREAM_RESPONSE': http_code = api_response.scrape_result['status_code'] if 'description' in api_response.error: description = api_response.error['description'] message = '%s %s %s' % (str(http_code), code, api_response.error['message']) if 'doc_url' in api_response.error: error_url = api_response.error['doc_url'] if 'retryable' in api_response.error: is_retryable = api_response.error['retryable'] resource = ErrorFactory._get_resource(code=code) if is_retryable is True: if 'X-Retry' in api_response.headers: retry_delay = int(api_response.headers['Retry-After']) message = '%s: %s' % (message, description) if description else message if retry_delay is not None and is_retryable is True: message = '%s. Retry delay : %s seconds' % (message, str(retry_delay)) args = { 'message': message, 'code': code, 'http_status_code': http_code, 'is_retryable': is_retryable, 'api_response': api_response, 'resource': resource, 'retry_delay': retry_delay, 'retry_times': retry_times, 'documentation_url': error_url, 'request': api_response.request, 'response': api_response.response } if kind == ScrapflyError.KIND_HTTP_BAD_RESPONSE: if http_code >= 500: return ApiHttpServerError(**args) is_scraper_api_error = resource in ErrorFactory.RESOURCE_TO_ERROR if http_code in ErrorFactory.HTTP_STATUS_TO_ERROR and not is_scraper_api_error: return ErrorFactory.HTTP_STATUS_TO_ERROR[http_code](**args) if is_scraper_api_error: return ErrorFactory.RESOURCE_TO_ERROR[resource](**args) return ApiHttpClientError(**args) elif kind == ScrapflyError.KIND_SCRAPFLY_ERROR: if code == 'ERR::SCRAPE::BAD_UPSTREAM_RESPONSE': if http_code >= 500: return UpstreamHttpServerError(**args) if http_code >= 400: return UpstreamHttpClientError(**args) if resource in ErrorFactory.RESOURCE_TO_ERROR: return ErrorFactory.RESOURCE_TO_ERROR[resource](**args) return ScrapflyError(**args)
class ExtractionAPIError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class ExtractionAPIError(HttpError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class ExtractionApiResponse (request: requests.models.Request,
response: requests.models.Response,
extraction_config: ExtractionConfig,
api_result: bytes | None = None)-
Expand source code
class ExtractionApiResponse(ApiResponse): def __init__(self, request: Request, response: Response, extraction_config: ExtractionConfig, api_result: Optional[bytes] = None): super().__init__(request, response) self.extraction_config = extraction_config self.result = self.handle_api_result(api_result) @property def extraction_result(self) -> Optional[Dict]: extraction_result = self.result.get('result', None) if not extraction_result: # handle empty extraction responses return {'data': None, 'content_type': None} else: return extraction_result @property def data(self) -> Union[Dict, List, str]: # depends on the LLM prompt if self.error is None: return self.extraction_result['data'] return None @property def content_type(self) -> Optional[str]: if self.error is None: return self.extraction_result['content_type'] return None @property def extraction_success(self) -> bool: extraction_result = self.extraction_result if extraction_result is None or extraction_result['data'] is None: return False return True @property def error(self) -> Optional[Dict]: if self.extraction_result is None: return self.result return None def _is_api_error(self, api_result: Dict) -> bool: if api_result is None: return True return 'error_id' in api_result def handle_api_result(self, api_result: bytes) -> FrozenDict: if self._is_api_error(api_result=api_result) is True: return FrozenDict(api_result) return FrozenDict({'result': api_result}) def raise_for_result(self, raise_on_upstream_error=True, error_class=ExtractionAPIError): super().raise_for_result(raise_on_upstream_error=raise_on_upstream_error, error_class=error_class)Ancestors
Instance variables
prop content_type : str | None-
Expand source code
@property def content_type(self) -> Optional[str]: if self.error is None: return self.extraction_result['content_type'] return None prop data : Dict | List | str-
Expand source code
@property def data(self) -> Union[Dict, List, str]: # depends on the LLM prompt if self.error is None: return self.extraction_result['data'] return None prop error : Dict | None-
Expand source code
@property def error(self) -> Optional[Dict]: if self.extraction_result is None: return self.result return None prop extraction_result : Dict | None-
Expand source code
@property def extraction_result(self) -> Optional[Dict]: extraction_result = self.result.get('result', None) if not extraction_result: # handle empty extraction responses return {'data': None, 'content_type': None} else: return extraction_result prop extraction_success : bool-
Expand source code
@property def extraction_success(self) -> bool: extraction_result = self.extraction_result if extraction_result is None or extraction_result['data'] is None: return False return True
Methods
def handle_api_result(self, api_result: bytes) ‑> FrozenDict-
Expand source code
def handle_api_result(self, api_result: bytes) -> FrozenDict: if self._is_api_error(api_result=api_result) is True: return FrozenDict(api_result) return FrozenDict({'result': api_result}) def raise_for_result(self,
raise_on_upstream_error=True,
error_class=scrapfly.errors.ExtractionAPIError)-
Expand source code
def raise_for_result(self, raise_on_upstream_error=True, error_class=ExtractionAPIError): super().raise_for_result(raise_on_upstream_error=raise_on_upstream_error, error_class=error_class)
Inherited members
class ExtractionConfig (body: str | bytes,
content_type: str,
url: str | None = None,
charset: str | None = None,
extraction_template: str | None = None,
extraction_ephemeral_template: Dict | None = None,
extraction_prompt: str | None = None,
extraction_model: str | None = None,
is_document_compressed: bool | None = None,
document_compression_format: CompressionFormat | None = None,
webhook: str | None = None,
timeout: int | None = None,
raise_on_upstream_error: bool = True,
template: str | None = None,
ephemeral_template: Dict | None = None)-
Expand source code
class ExtractionConfig(BaseApiConfig): body: Union[str, bytes] content_type: str url: Optional[str] = None charset: Optional[str] = None extraction_template: Optional[str] = None # a saved template name extraction_ephemeral_template: Optional[Dict] # ephemeraly declared json template extraction_prompt: Optional[str] = None extraction_model: Optional[str] = None is_document_compressed: Optional[bool] = None document_compression_format: Optional[CompressionFormat] = None webhook: Optional[str] = None timeout: Optional[int] = None raise_on_upstream_error: bool = True # deprecated options template: Optional[str] = None ephemeral_template: Optional[Dict] = None def __init__( self, body: Union[str, bytes], content_type: str, url: Optional[str] = None, charset: Optional[str] = None, extraction_template: Optional[str] = None, # a saved template name extraction_ephemeral_template: Optional[Dict] = None, # ephemeraly declared json template extraction_prompt: Optional[str] = None, extraction_model: Optional[str] = None, is_document_compressed: Optional[bool] = None, document_compression_format: Optional[CompressionFormat] = None, webhook: Optional[str] = None, timeout: Optional[int] = None, raise_on_upstream_error: bool = True, # deprecated options template: Optional[str] = None, ephemeral_template: Optional[Dict] = None ): if template: warnings.warn( "Deprecation warning: 'template' is deprecated. Use 'extraction_template' instead." ) extraction_template = template if ephemeral_template: warnings.warn( "Deprecation warning: 'ephemeral_template' is deprecated. Use 'extraction_ephemeral_template' instead." ) extraction_ephemeral_template = ephemeral_template self.key = None self.body = body self.content_type = content_type self.url = url self.charset = charset self.extraction_template = extraction_template self.extraction_ephemeral_template = extraction_ephemeral_template self.extraction_prompt = extraction_prompt self.extraction_model = extraction_model self.is_document_compressed = is_document_compressed self.document_compression_format = CompressionFormat(document_compression_format) if document_compression_format else None self.webhook = webhook self.timeout = timeout self.raise_on_upstream_error = raise_on_upstream_error if isinstance(body, bytes) or document_compression_format: compression_format = detect_compression_format(body) if compression_format is not None: self.is_document_compressed = True if self.document_compression_format and compression_format != self.document_compression_format: raise ExtractionConfigError( f'The detected compression format `{compression_format}` does not match declared format `{self.document_compression_format}`. ' f'You must pass the compression format or disable compression.' ) self.document_compression_format = compression_format else: self.is_document_compressed = False if self.is_document_compressed is False: compression_foramt = CompressionFormat(self.document_compression_format) if self.document_compression_format else None if isinstance(self.body, str) and compression_foramt: self.body = self.body.encode('utf-8') if compression_foramt == CompressionFormat.GZIP: import gzip self.body = gzip.compress(self.body) elif compression_foramt == CompressionFormat.ZSTD: try: import zstandard as zstd except ImportError: raise ExtractionConfigError( f'zstandard is not installed. You must run pip install zstandard' f' to auto compress into zstd or use compression formats.' ) self.body = zstd.compress(self.body) elif compression_foramt == CompressionFormat.DEFLATE: import zlib compressor = zlib.compressobj(wbits=-zlib.MAX_WBITS) # raw deflate compression self.body = compressor.compress(self.body) + compressor.flush() def to_api_params(self, key: str) -> Dict: params = { 'key': self.key or key, 'content_type': self.content_type } if self.url: params['url'] = self.url if self.charset: params['charset'] = self.charset if self.extraction_template and self.extraction_ephemeral_template: raise ExtractionConfigError('You cannot pass both parameters extraction_template and extraction_ephemeral_template. You must choose') if self.extraction_template: params['extraction_template'] = 'persistent:' + self.extraction_template if self.extraction_ephemeral_template: template_json = json.dumps(self.extraction_ephemeral_template) params['extraction_template'] = 'ephemeral:' + urlsafe_b64encode(template_json.encode('utf-8')).decode('utf-8') if self.extraction_prompt: params['extraction_prompt'] = quote_plus(self.extraction_prompt) if self.extraction_model: params['extraction_model'] = self.extraction_model if self.webhook: params['webhook_name'] = self.webhook if self.timeout is not None: params['timeout'] = self.timeout return params def to_dict(self) -> Dict: """ Export the ExtractionConfig instance to a plain dictionary. """ if self.is_document_compressed is True: compression_foramt = CompressionFormat(self.document_compression_format) if self.document_compression_format else None if compression_foramt == CompressionFormat.GZIP: import gzip self.body = gzip.decompress(self.body) elif compression_foramt == CompressionFormat.ZSTD: import zstandard as zstd self.body = zstd.decompress(self.body) elif compression_foramt == CompressionFormat.DEFLATE: import zlib decompressor = zlib.decompressobj(wbits=-zlib.MAX_WBITS) self.body = decompressor.decompress(self.body) + decompressor.flush() if isinstance(self.body, bytes): self.body = self.body.decode('utf-8') self.is_document_compressed = False return { 'body': self.body, 'content_type': self.content_type, 'url': self.url, 'charset': self.charset, 'extraction_template': self.extraction_template, 'extraction_ephemeral_template': self.extraction_ephemeral_template, 'extraction_prompt': self.extraction_prompt, 'extraction_model': self.extraction_model, 'is_document_compressed': self.is_document_compressed, 'document_compression_format': CompressionFormat(self.document_compression_format).value if self.document_compression_format else None, 'webhook': self.webhook, 'raise_on_upstream_error': self.raise_on_upstream_error, } @staticmethod def from_dict(extraction_config_dict: Dict) -> 'ExtractionConfig': """Create an ExtractionConfig instance from a dictionary.""" body = extraction_config_dict.get('body', None) content_type = extraction_config_dict.get('content_type', None) url = extraction_config_dict.get('url', None) charset = extraction_config_dict.get('charset', None) extraction_template = extraction_config_dict.get('extraction_template', None) extraction_ephemeral_template = extraction_config_dict.get('extraction_ephemeral_template', None) extraction_prompt = extraction_config_dict.get('extraction_prompt', None) extraction_model = extraction_config_dict.get('extraction_model', None) is_document_compressed = extraction_config_dict.get('is_document_compressed', None) document_compression_format = extraction_config_dict.get('document_compression_format', None) document_compression_format = CompressionFormat(document_compression_format) if document_compression_format else None webhook = extraction_config_dict.get('webhook', None) raise_on_upstream_error = extraction_config_dict.get('raise_on_upstream_error', True) return ExtractionConfig( body=body, content_type=content_type, url=url, charset=charset, extraction_template=extraction_template, extraction_ephemeral_template=extraction_ephemeral_template, extraction_prompt=extraction_prompt, extraction_model=extraction_model, is_document_compressed=is_document_compressed, document_compression_format=document_compression_format, webhook=webhook, raise_on_upstream_error=raise_on_upstream_error )Ancestors
Class variables
var body : str | bytesvar charset : str | Nonevar content_type : strvar document_compression_format : CompressionFormat | Nonevar ephemeral_template : Dict | Nonevar extraction_ephemeral_template : Dict | Nonevar extraction_model : str | Nonevar extraction_prompt : str | Nonevar extraction_template : str | Nonevar is_document_compressed : bool | Nonevar raise_on_upstream_error : boolvar template : str | Nonevar timeout : int | Nonevar url : str | Nonevar webhook : str | None
Static methods
def from_dict(extraction_config_dict: Dict) ‑> ExtractionConfig-
Expand source code
@staticmethod def from_dict(extraction_config_dict: Dict) -> 'ExtractionConfig': """Create an ExtractionConfig instance from a dictionary.""" body = extraction_config_dict.get('body', None) content_type = extraction_config_dict.get('content_type', None) url = extraction_config_dict.get('url', None) charset = extraction_config_dict.get('charset', None) extraction_template = extraction_config_dict.get('extraction_template', None) extraction_ephemeral_template = extraction_config_dict.get('extraction_ephemeral_template', None) extraction_prompt = extraction_config_dict.get('extraction_prompt', None) extraction_model = extraction_config_dict.get('extraction_model', None) is_document_compressed = extraction_config_dict.get('is_document_compressed', None) document_compression_format = extraction_config_dict.get('document_compression_format', None) document_compression_format = CompressionFormat(document_compression_format) if document_compression_format else None webhook = extraction_config_dict.get('webhook', None) raise_on_upstream_error = extraction_config_dict.get('raise_on_upstream_error', True) return ExtractionConfig( body=body, content_type=content_type, url=url, charset=charset, extraction_template=extraction_template, extraction_ephemeral_template=extraction_ephemeral_template, extraction_prompt=extraction_prompt, extraction_model=extraction_model, is_document_compressed=is_document_compressed, document_compression_format=document_compression_format, webhook=webhook, raise_on_upstream_error=raise_on_upstream_error )Create an ExtractionConfig instance from a dictionary.
Methods
def to_api_params(self, key: str) ‑> Dict-
Expand source code
def to_api_params(self, key: str) -> Dict: params = { 'key': self.key or key, 'content_type': self.content_type } if self.url: params['url'] = self.url if self.charset: params['charset'] = self.charset if self.extraction_template and self.extraction_ephemeral_template: raise ExtractionConfigError('You cannot pass both parameters extraction_template and extraction_ephemeral_template. You must choose') if self.extraction_template: params['extraction_template'] = 'persistent:' + self.extraction_template if self.extraction_ephemeral_template: template_json = json.dumps(self.extraction_ephemeral_template) params['extraction_template'] = 'ephemeral:' + urlsafe_b64encode(template_json.encode('utf-8')).decode('utf-8') if self.extraction_prompt: params['extraction_prompt'] = quote_plus(self.extraction_prompt) if self.extraction_model: params['extraction_model'] = self.extraction_model if self.webhook: params['webhook_name'] = self.webhook if self.timeout is not None: params['timeout'] = self.timeout return params def to_dict(self) ‑> Dict-
Expand source code
def to_dict(self) -> Dict: """ Export the ExtractionConfig instance to a plain dictionary. """ if self.is_document_compressed is True: compression_foramt = CompressionFormat(self.document_compression_format) if self.document_compression_format else None if compression_foramt == CompressionFormat.GZIP: import gzip self.body = gzip.decompress(self.body) elif compression_foramt == CompressionFormat.ZSTD: import zstandard as zstd self.body = zstd.decompress(self.body) elif compression_foramt == CompressionFormat.DEFLATE: import zlib decompressor = zlib.decompressobj(wbits=-zlib.MAX_WBITS) self.body = decompressor.decompress(self.body) + decompressor.flush() if isinstance(self.body, bytes): self.body = self.body.decode('utf-8') self.is_document_compressed = False return { 'body': self.body, 'content_type': self.content_type, 'url': self.url, 'charset': self.charset, 'extraction_template': self.extraction_template, 'extraction_ephemeral_template': self.extraction_ephemeral_template, 'extraction_prompt': self.extraction_prompt, 'extraction_model': self.extraction_model, 'is_document_compressed': self.is_document_compressed, 'document_compression_format': CompressionFormat(self.document_compression_format).value if self.document_compression_format else None, 'webhook': self.webhook, 'raise_on_upstream_error': self.raise_on_upstream_error, }Export the ExtractionConfig instance to a plain dictionary.
class HarArchive (har_data: bytes)-
Expand source code
class HarArchive: """Parser and accessor for HAR (HTTP Archive) format data""" def __init__(self, har_data: bytes): """ Initialize HAR archive from bytes Args: har_data: HAR file content as bytes (JSON format, may be gzipped) """ # Decompress if gzipped if isinstance(har_data, bytes): if har_data[:2] == b'\x1f\x8b': # gzip magic number har_data = gzip.decompress(har_data) har_data = har_data.decode('utf-8') # Parse the special format: {"log":{...,"entries":[]}}{"entry1"}{"entry2"}... # First object is HAR log structure, subsequent objects are individual entries objects = [] decoder = json.JSONDecoder() idx = 0 while idx < len(har_data): har_data_stripped = har_data[idx:].lstrip() if not har_data_stripped: break try: obj, end_idx = decoder.raw_decode(har_data_stripped) objects.append(obj) idx += len(har_data[idx:]) - len(har_data_stripped) + end_idx except json.JSONDecodeError: break # First object should be the HAR log structure if objects and 'log' in objects[0]: self._data = objects[0] self._log = self._data.get('log', {}) # Remaining objects are the entries self._entries = objects[1:] if len(objects) > 1 else [] else: # Fallback: standard HAR format self._data = json.loads(har_data) if isinstance(har_data, str) else {} self._log = self._data.get('log', {}) self._entries = self._log.get('entries', []) @property def version(self) -> str: """Get HAR version""" return self._log.get('version', '') @property def creator(self) -> Dict[str, Any]: """Get creator information""" return self._log.get('creator', {}) @property def pages(self) -> List[Dict[str, Any]]: """Get pages list""" return self._log.get('pages', []) def get_entries(self) -> List[HarEntry]: """ Get all entries as list Returns: List of HarEntry objects """ return [HarEntry(entry) for entry in self._entries] def iter_entries(self) -> Iterator[HarEntry]: """ Iterate through all HAR entries Yields: HarEntry objects """ for entry in self._entries: yield HarEntry(entry) def get_urls(self) -> List[str]: """ Get all URLs in the archive Returns: List of unique URLs """ urls = [] for entry in self._entries: url = entry.get('request', {}).get('url', '') if url and url not in urls: urls.append(url) return urls def find_by_url(self, url: str) -> Optional[HarEntry]: """ Find entry by exact URL match Args: url: URL to search for Returns: First matching HarEntry or None """ for entry in self.iter_entries(): if entry.url == url: return entry return None def filter_by_status(self, status_code: int) -> List[HarEntry]: """ Filter entries by status code Args: status_code: HTTP status code to filter by Returns: List of matching HarEntry objects """ return [entry for entry in self.iter_entries() if entry.status_code == status_code] def filter_by_content_type(self, content_type: str) -> List[HarEntry]: """ Filter entries by content type (substring match) Args: content_type: Content type to filter by (e.g., 'text/html') Returns: List of matching HarEntry objects """ return [entry for entry in self.iter_entries() if content_type.lower() in entry.content_type.lower()] def __len__(self) -> int: """Get number of entries""" return len(self._entries) def __repr__(self) -> str: return f"<HarArchive {len(self._entries)} entries>"Parser and accessor for HAR (HTTP Archive) format data
Initialize HAR archive from bytes
Args
har_data- HAR file content as bytes (JSON format, may be gzipped)
Instance variables
prop creator : Dict[str, Any]-
Expand source code
@property def creator(self) -> Dict[str, Any]: """Get creator information""" return self._log.get('creator', {})Get creator information
prop pages : List[Dict[str, Any]]-
Expand source code
@property def pages(self) -> List[Dict[str, Any]]: """Get pages list""" return self._log.get('pages', [])Get pages list
prop version : str-
Expand source code
@property def version(self) -> str: """Get HAR version""" return self._log.get('version', '')Get HAR version
Methods
def filter_by_content_type(self, content_type: str) ‑> List[HarEntry]-
Expand source code
def filter_by_content_type(self, content_type: str) -> List[HarEntry]: """ Filter entries by content type (substring match) Args: content_type: Content type to filter by (e.g., 'text/html') Returns: List of matching HarEntry objects """ return [entry for entry in self.iter_entries() if content_type.lower() in entry.content_type.lower()]Filter entries by content type (substring match)
Args
content_type- Content type to filter by (e.g., 'text/html')
Returns
List of matching HarEntry objects
def filter_by_status(self, status_code: int) ‑> List[HarEntry]-
Expand source code
def filter_by_status(self, status_code: int) -> List[HarEntry]: """ Filter entries by status code Args: status_code: HTTP status code to filter by Returns: List of matching HarEntry objects """ return [entry for entry in self.iter_entries() if entry.status_code == status_code]Filter entries by status code
Args
status_code- HTTP status code to filter by
Returns
List of matching HarEntry objects
def find_by_url(self, url: str) ‑> HarEntry | None-
Expand source code
def find_by_url(self, url: str) -> Optional[HarEntry]: """ Find entry by exact URL match Args: url: URL to search for Returns: First matching HarEntry or None """ for entry in self.iter_entries(): if entry.url == url: return entry return NoneFind entry by exact URL match
Args
url- URL to search for
Returns
First matching HarEntry or None
def get_entries(self) ‑> List[HarEntry]-
Expand source code
def get_entries(self) -> List[HarEntry]: """ Get all entries as list Returns: List of HarEntry objects """ return [HarEntry(entry) for entry in self._entries]Get all entries as list
Returns
List of HarEntry objects
def get_urls(self) ‑> List[str]-
Expand source code
def get_urls(self) -> List[str]: """ Get all URLs in the archive Returns: List of unique URLs """ urls = [] for entry in self._entries: url = entry.get('request', {}).get('url', '') if url and url not in urls: urls.append(url) return urlsGet all URLs in the archive
Returns
List of unique URLs
def iter_entries(self) ‑> Iterator[HarEntry]-
Expand source code
def iter_entries(self) -> Iterator[HarEntry]: """ Iterate through all HAR entries Yields: HarEntry objects """ for entry in self._entries: yield HarEntry(entry)Iterate through all HAR entries
Yields
HarEntry objects
class HarEntry (entry_data: Dict[str, Any])-
Expand source code
class HarEntry: """Represents a single HAR entry (HTTP request/response pair)""" def __init__(self, entry_data: Dict[str, Any]): """ Initialize from HAR entry dict Args: entry_data: HAR entry dictionary """ self._data = entry_data self._request = entry_data.get('request', {}) self._response = entry_data.get('response', {}) @property def url(self) -> str: """Get request URL""" return self._request.get('url', '') @property def method(self) -> str: """Get HTTP method""" return self._request.get('method', 'GET') @property def status_code(self) -> int: """Get response status code""" # Handle case where response doesn't exist or status is missing if not self._response: return 0 status = self._response.get('status') if status is None: return 0 # Ensure it's an int (HAR data might have status as string) try: return int(status) except (ValueError, TypeError): return 0 @property def status_text(self) -> str: """Get response status text""" return self._response.get('statusText', '') @property def request_headers(self) -> Dict[str, str]: """Get request headers as dict""" headers = {} for header in self._request.get('headers', []): headers[header['name']] = header['value'] return headers @property def response_headers(self) -> Dict[str, str]: """Get response headers as dict""" headers = {} for header in self._response.get('headers', []): headers[header['name']] = header['value'] return headers @property def content(self) -> bytes: """Get response content as bytes""" content_data = self._response.get('content', {}) text = content_data.get('text', '') # Handle base64 encoding if present encoding = content_data.get('encoding', '') if encoding == 'base64': import base64 return base64.b64decode(text) # Return as UTF-8 bytes if isinstance(text, str): return text.encode('utf-8') return text @property def content_type(self) -> str: """Get response content type""" return self._response.get('content', {}).get('mimeType', '') @property def content_size(self) -> int: """Get response content size""" return self._response.get('content', {}).get('size', 0) @property def started_datetime(self) -> str: """Get when request was started (ISO 8601 format)""" return self._data.get('startedDateTime', '') @property def time(self) -> float: """Get total elapsed time in milliseconds""" return self._data.get('time', 0.0) @property def timings(self) -> Dict[str, float]: """Get detailed timing information""" return self._data.get('timings', {}) def __repr__(self) -> str: return f"<HarEntry {self.method} {self.url} [{self.status_code}]>"Represents a single HAR entry (HTTP request/response pair)
Initialize from HAR entry dict
Args
entry_data- HAR entry dictionary
Instance variables
prop content : bytes-
Expand source code
@property def content(self) -> bytes: """Get response content as bytes""" content_data = self._response.get('content', {}) text = content_data.get('text', '') # Handle base64 encoding if present encoding = content_data.get('encoding', '') if encoding == 'base64': import base64 return base64.b64decode(text) # Return as UTF-8 bytes if isinstance(text, str): return text.encode('utf-8') return textGet response content as bytes
prop content_size : int-
Expand source code
@property def content_size(self) -> int: """Get response content size""" return self._response.get('content', {}).get('size', 0)Get response content size
prop content_type : str-
Expand source code
@property def content_type(self) -> str: """Get response content type""" return self._response.get('content', {}).get('mimeType', '')Get response content type
prop method : str-
Expand source code
@property def method(self) -> str: """Get HTTP method""" return self._request.get('method', 'GET')Get HTTP method
prop request_headers : Dict[str, str]-
Expand source code
@property def request_headers(self) -> Dict[str, str]: """Get request headers as dict""" headers = {} for header in self._request.get('headers', []): headers[header['name']] = header['value'] return headersGet request headers as dict
prop response_headers : Dict[str, str]-
Expand source code
@property def response_headers(self) -> Dict[str, str]: """Get response headers as dict""" headers = {} for header in self._response.get('headers', []): headers[header['name']] = header['value'] return headersGet response headers as dict
prop started_datetime : str-
Expand source code
@property def started_datetime(self) -> str: """Get when request was started (ISO 8601 format)""" return self._data.get('startedDateTime', '')Get when request was started (ISO 8601 format)
prop status_code : int-
Expand source code
@property def status_code(self) -> int: """Get response status code""" # Handle case where response doesn't exist or status is missing if not self._response: return 0 status = self._response.get('status') if status is None: return 0 # Ensure it's an int (HAR data might have status as string) try: return int(status) except (ValueError, TypeError): return 0Get response status code
prop status_text : str-
Expand source code
@property def status_text(self) -> str: """Get response status text""" return self._response.get('statusText', '')Get response status text
prop time : float-
Expand source code
@property def time(self) -> float: """Get total elapsed time in milliseconds""" return self._data.get('time', 0.0)Get total elapsed time in milliseconds
prop timings : Dict[str, float]-
Expand source code
@property def timings(self) -> Dict[str, float]: """Get detailed timing information""" return self._data.get('timings', {})Get detailed timing information
prop url : str-
Expand source code
@property def url(self) -> str: """Get request URL""" return self._request.get('url', '')Get request URL
class HttpError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class HttpError(ScrapflyError): def __init__(self, request:Request, response:Optional[Response]=None, **kwargs): self.request = request self.response = response super().__init__(**kwargs) def __str__(self) -> str: if isinstance(self, UpstreamHttpError): return f"Target website responded with {self.api_response.scrape_result['status_code']} - {self.api_response.scrape_result['reason']}" if self.api_response is not None: return self.api_response.error_message text = f"{self.response.status_code} - {self.response.reason}" # Include detailed error message for all HTTP errors if self.message: text += f" - {self.message}" return textCommon base class for all non-exit exceptions.
Ancestors
- ScrapflyError
- builtins.Exception
- builtins.BaseException
Subclasses
- ApiHttpClientError
- scrapfly.errors.ExtractionAPIError
- scrapfly.errors.QuotaLimitReached
- scrapfly.errors.ScraperAPIError
- scrapfly.errors.ScreenshotAPIError
- scrapfly.errors.TooManyConcurrentRequest
- scrapfly.errors.UpstreamHttpError
class ListSchedulesOptions (kind: str | None = None, status: str | None = None)-
Expand source code
@dataclass class ListSchedulesOptions: """Filter options for list_schedules / list_<kind>_schedules. Use either this dataclass or the equivalent keyword arguments interchangeably.""" kind: Optional[str] = None # "api.scrape" | "api.screenshot" | "api.crawler" status: Optional[str] = None # "ACTIVE" | "PAUSED" | "CANCELLED"Filter options for list_schedules / list_
_schedules. Use either this dataclass or the equivalent keyword arguments interchangeably. Instance variables
var kind : str | Nonevar status : str | None
class OperatingSystem (value, names=None, *, module=None, qualname=None, type=None, start=1)-
Expand source code
class OperatingSystem(Enum): LINUX = "linux" WINDOWS = "windows" MACOS = "macos" ANDROID = "android" IPHONE = "iphone" IPAD = "ipad"An enumeration.
Ancestors
- enum.Enum
Class variables
var ANDROIDvar IPADvar IPHONEvar LINUXvar MACOSvar WINDOWS
class ProxyPool (value, names=None, *, module=None, qualname=None, type=None, start=1)-
Expand source code
class ProxyPool(Enum): DATACENTER = "datacenter" RESIDENTIAL = "residential"An enumeration.
Ancestors
- enum.Enum
Class variables
var DATACENTERvar RESIDENTIAL
class ResponseBodyHandler (use_brotli: bool = False,
signing_secrets: str | Tuple[str, ...] | None = None)-
Expand source code
class ResponseBodyHandler: SUPPORTED_COMPRESSION = ['gzip', 'deflate'] SUPPORTED_CONTENT_TYPES = ['application/msgpack', 'application/json'] class JSONDateTimeDecoder(JSONDecoder): def __init__(self, *args, **kargs): JSONDecoder.__init__(self, *args, object_hook=_date_parser, **kargs) # brotli under perform at same gzip level and upper level destroy the cpu so # the trade off do not worth it for most of usage def __init__(self, use_brotli: bool = False, signing_secrets: Optional[Union[str, Tuple[str, ...]]] = None): if use_brotli is True and 'br' not in self.SUPPORTED_COMPRESSION: try: try: import brotlicffi as brotli self.SUPPORTED_COMPRESSION.insert(0, 'br') except ImportError: import brotli self.SUPPORTED_COMPRESSION.insert(0, 'br') except ImportError: pass try: from urllib3.response import HAS_ZSTD if HAS_ZSTD and 'zstd' not in self.SUPPORTED_COMPRESSION: self.SUPPORTED_COMPRESSION.append('zstd') except ImportError: pass self.content_encoding: str = ', '.join(self.SUPPORTED_COMPRESSION) self._signing_secret: Optional[Tuple[str]] = None if signing_secrets is not None: # A bare str is iterable: without this it yields one HMAC key per character. if isinstance(signing_secrets, (str, bytes)): signing_secrets = (signing_secrets,) _secrets = set() for signing_secret in signing_secrets: if not isinstance(signing_secret, str) or not signing_secret: raise ValueError('signing_secrets must be non-empty strings, as shown in the webhook dashboard') _secrets.add(signing_secret.encode('utf-8')) if not _secrets: raise ValueError('signing_secrets was empty; omit it entirely to build a handler that does not verify') self._signing_secret = tuple(_secrets) try: # automatically use msgpack if available https://msgpack.org/ import msgpack self.accept = 'application/msgpack;charset=utf-8' self.content_type = 'application/msgpack;charset=utf-8' self.content_loader = partial(msgpack.loads, object_hook=_date_parser, strict_map_key=False) except ImportError: self.accept = 'application/json;charset=utf-8' self.content_type = 'application/json;charset=utf-8' self.content_loader = partial(loads, cls=self.JSONDateTimeDecoder) def support(self, headers: Dict) -> bool: if 'content-type' not in headers: return False for content_type in self.SUPPORTED_CONTENT_TYPES: if headers['content-type'].find(content_type) != -1: return True return False def verify(self, message: bytes, signature: Optional[str]) -> bool: if self._signing_secret is None: raise ValueError('no signing secret configured; pass signing_secrets to ResponseBodyHandler') # An absent header is a failed verification, not a programming error. if not signature: return False # The digest is sent as upper hex; the -Lowercase header carries the same value. received = signature.strip().upper().encode('utf-8') for signing_secret in self._signing_secret: computed = hmac.new(signing_secret, message, hashlib.sha256).hexdigest().upper().encode('ascii') if hmac.compare_digest(computed, received): return True return False def read( self, content: bytes, content_encoding: str, content_type: str, signature: Optional[str], max_decompressed_size: int = MAX_DECOMPRESSED_SIZE, signature_message: Optional[bytes] = None, ) -> Dict: # Signing happens before Content-Encoding is applied, so the body is inflated # before it can be verified. Bound the output: this runs pre-authentication. content = decompress(content, content_encoding, max_decompressed_size) # Gate on the secret being configured, never on the header being supplied: # an absent signature is a forgery, not an exemption. if self._signing_secret is not None: if not self.verify(content if signature_message is None else signature_message, signature): raise WebhookSignatureMissMatch() content_type = content_type or '' if content_type.startswith('application/json'): content = loads(content, cls=self.JSONDateTimeDecoder) elif content_type.startswith('application/msgpack'): import msgpack content = msgpack.loads(content, object_hook=_date_parser, strict_map_key=False) return content def __call__(self, content: bytes, content_type: str) -> Union[str, Dict]: content_loader = None if content_type.find('application/json') != -1: content_loader = partial(loads, cls=self.JSONDateTimeDecoder) elif content_type.find('application/msgpack') != -1: import msgpack content_loader = partial(msgpack.loads, object_hook=_date_parser, strict_map_key=False) if content_loader is None: raise Exception('Unsupported content type') try: return content_loader(content) except Exception as e: try: raise EncoderError(content=content.decode('utf-8')) from e except UnicodeError: raise EncoderError(content=base64.b64encode(content).decode('utf-8')) from eClass variables
var JSONDateTimeDecoder-
Simple JSON https://json.org decoder
Performs the following translations in decoding by default:
+---------------+-------------------+ | JSON | Python | +===============+===================+ | object | dict | +---------------+-------------------+ | array | list | +---------------+-------------------+ | string | str | +---------------+-------------------+ | number (int) | int | +---------------+-------------------+ | number (real) | float | +---------------+-------------------+ | true | True | +---------------+-------------------+ | false | False | +---------------+-------------------+ | null | None | +---------------+-------------------+
It also understands
NaN,Infinity, and-Infinityas their correspondingfloatvalues, which is outside the JSON spec. var SUPPORTED_COMPRESSIONvar SUPPORTED_CONTENT_TYPES
Methods
def read(self,
content: bytes,
content_encoding: str,
content_type: str,
signature: str | None,
max_decompressed_size: int = 268435456,
signature_message: bytes | None = None) ‑> Dict-
Expand source code
def read( self, content: bytes, content_encoding: str, content_type: str, signature: Optional[str], max_decompressed_size: int = MAX_DECOMPRESSED_SIZE, signature_message: Optional[bytes] = None, ) -> Dict: # Signing happens before Content-Encoding is applied, so the body is inflated # before it can be verified. Bound the output: this runs pre-authentication. content = decompress(content, content_encoding, max_decompressed_size) # Gate on the secret being configured, never on the header being supplied: # an absent signature is a forgery, not an exemption. if self._signing_secret is not None: if not self.verify(content if signature_message is None else signature_message, signature): raise WebhookSignatureMissMatch() content_type = content_type or '' if content_type.startswith('application/json'): content = loads(content, cls=self.JSONDateTimeDecoder) elif content_type.startswith('application/msgpack'): import msgpack content = msgpack.loads(content, object_hook=_date_parser, strict_map_key=False) return content def support(self, headers: Dict) ‑> bool-
Expand source code
def support(self, headers: Dict) -> bool: if 'content-type' not in headers: return False for content_type in self.SUPPORTED_CONTENT_TYPES: if headers['content-type'].find(content_type) != -1: return True return False def verify(self, message: bytes, signature: str | None) ‑> bool-
Expand source code
def verify(self, message: bytes, signature: Optional[str]) -> bool: if self._signing_secret is None: raise ValueError('no signing secret configured; pass signing_secrets to ResponseBodyHandler') # An absent header is a failed verification, not a programming error. if not signature: return False # The digest is sent as upper hex; the -Lowercase header carries the same value. received = signature.strip().upper().encode('utf-8') for signing_secret in self._signing_secret: computed = hmac.new(signing_secret, message, hashlib.sha256).hexdigest().upper().encode('ascii') if hmac.compare_digest(computed, received): return True return False
class ScheduleAPIError (message: str, code: str, http_status_code: int, details: Any = None)-
Expand source code
class ScheduleAPIError(Exception): """Raised on any non-2xx response from a /schedules/* endpoint. The ``code`` attribute carries the public ``ERR::SCHEDULER::*`` identifier so callers can branch on it without parsing the message string. """ def __init__( self, message: str, code: str, http_status_code: int, details: Any = None, ) -> None: super().__init__(message) self.code = code self.http_status_code = http_status_code self.details = details def __str__(self) -> str: # noqa: D401 return f"{self.code} ({self.http_status_code}): {self.args[0] if self.args else ''}"Raised on any non-2xx response from a /schedules/* endpoint.
The
codeattribute carries the publicERR::SCHEDULER::*identifier so callers can branch on it without parsing the message string.Ancestors
- builtins.Exception
- builtins.BaseException
class ScheduleEnd (type: str, date: str | None = None, count: int | None = None)-
Expand source code
@dataclass class ScheduleEnd: """Bounds a recurring schedule by either a date or a fire count.""" type: str # "date" | "count" date: Optional[str] = None count: Optional[int] = NoneBounds a recurring schedule by either a date or a fire count.
Instance variables
var count : int | Nonevar date : str | Nonevar type : str
class ScheduleRecurrence (cron: str | None = None,
interval: int | None = None,
unit: str | None = None,
days: List[str] | None = None,
ends: ScheduleEnd | None = None)-
Expand source code
@dataclass class ScheduleRecurrence: """When a schedule fires next. Cron mode wins when ``cron`` is set; otherwise ``interval`` + ``unit`` drive the cadence. All times are interpreted in UTC server-side. """ cron: Optional[str] = None interval: Optional[int] = None unit: Optional[str] = None # "minute" | "hour" | "day" | "week" | "month" days: Optional[List[str]] = None ends: Optional[ScheduleEnd] = None def to_dict(self) -> Dict[str, Any]: out: Dict[str, Any] = {} if self.cron: out["cron"] = self.cron if self.interval is not None: out["interval"] = self.interval if self.unit: out["unit"] = self.unit if self.days: out["days"] = self.days if self.ends: ends: Dict[str, Any] = {"type": self.ends.type} if self.ends.date: ends["date"] = self.ends.date if self.ends.count is not None: ends["count"] = self.ends.count out["ends"] = ends return outWhen a schedule fires next.
Cron mode wins when
cronis set; otherwiseinterval+unitdrive the cadence. All times are interpreted in UTC server-side.Instance variables
var cron : str | Nonevar days : List[str] | Nonevar ends : ScheduleEnd | Nonevar interval : int | Nonevar unit : str | None
Methods
def to_dict(self) ‑> Dict[str, Any]-
Expand source code
def to_dict(self) -> Dict[str, Any]: out: Dict[str, Any] = {} if self.cron: out["cron"] = self.cron if self.interval is not None: out["interval"] = self.interval if self.unit: out["unit"] = self.unit if self.days: out["days"] = self.days if self.ends: ends: Dict[str, Any] = {"type": self.ends.type} if self.ends.date: ends["date"] = self.ends.date if self.ends.count is not None: ends["count"] = self.ends.count out["ends"] = ends return out
class ScrapeApiResponse (request: requests.models.Request,
response: requests.models.Response,
scrape_config: ScrapeConfig,
api_result: Dict | None = None,
large_object_handler: Callable | None = None)-
Expand source code
class ScrapeApiResponse(ApiResponse): scrape_config:ScrapeConfig large_object_handler:Callable def __init__(self, request: Request, response: Response, scrape_config: ScrapeConfig, api_result: Optional[Dict] = None, large_object_handler:Optional[Callable]=None): super().__init__(request, response) self.scrape_config = scrape_config self.large_object_handler = large_object_handler if self.scrape_config.method == 'HEAD': api_result = { 'result': { 'request_headers': {}, 'status': 'DONE', 'success': 200 <= self.response.status_code < 300, 'response_headers': self.response.headers, 'status_code': self.response.status_code, 'reason': self.response.reason, 'format': 'text', 'content': '' }, 'context': {}, 'config': self.scrape_config.__dict__ } if 'X-Scrapfly-Reject-Code' in self.response.headers: api_result['result']['error'] = { 'code': self.response.headers['X-Scrapfly-Reject-Code'], 'http_code': int(self.response.headers['X-Scrapfly-Reject-Http-Code']), 'message': self.response.headers['X-Scrapfly-Reject-Description'], 'error_id': self.response.headers['X-Scrapfly-Reject-ID'], 'retryable': True if self.response.headers['X-Scrapfly-Reject-Retryable'] == 'yes' else False, 'doc_url': '', 'links': {} } if 'X-Scrapfly-Reject-Doc' in self.response.headers: api_result['result']['error']['doc_url'] = self.response.headers['X-Scrapfly-Reject-Doc'] api_result['result']['error']['links']['Related Docs'] = self.response.headers['X-Scrapfly-Reject-Doc'] if isinstance(api_result, str): raise HttpError( request=request, response=response, message='Bad gateway', code=502, http_status_code=502, is_retryable=True ) self.result = self.handle_api_result(api_result=api_result) @property def scrape_result(self) -> Optional[Dict]: return self.result.get('result', None) @property def config(self) -> Optional[Dict]: if self.scrape_result is None: return None return self.result['config'] @property def context(self) -> Optional[Dict]: if self.scrape_result is None: return None return self.result['context'] @property def content(self) -> str: if self.scrape_result is None: return '' return self.scrape_result['content'] @property def success(self) -> bool: """ Success means Scrapfly api reply correctly to the call, but the scrape can be unsuccessful if the upstream reply with error status code """ return 200 <= self.response.status_code <= 299 @property def scrape_success(self) -> bool: scrape_result = self.scrape_result if not scrape_result: return False return self.scrape_result['success'] @property def error(self) -> Optional[Dict]: if self.scrape_result is None: return None if self.scrape_success is False: return self.scrape_result.get('error') @property def upstream_status_code(self) -> Optional[int]: if self.scrape_result is None: return None if 'status_code' in self.scrape_result: return self.scrape_result['status_code'] return None @cached_property def soup(self) -> 'BeautifulSoup': if self.scrape_result['format'] != 'text': raise ContentError("Unable to cast into beautiful soup, the format of data is binary - must be text content") try: from bs4 import BeautifulSoup soup = BeautifulSoup(self.content, "lxml") return soup except ImportError as e: logger.error('You must install scrapfly[parser] to enable this feature') @cached_property def selector(self) -> 'Selector': if self.scrape_result['format'] != 'text': raise ContentError("Unable to cast into beautiful soup, the format of data is binary - must be text content") try: from parsel import Selector return Selector(text=self.content) except ImportError as e: logger.error('You must install parsel or scrapy package to enable this feature') raise e def handle_api_result(self, api_result: Dict) -> Optional[FrozenDict]: if self._is_api_error(api_result=api_result) is True: return FrozenDict(api_result) try: if isinstance(api_result['config']['headers'], list): api_result['config']['headers'] = {} except TypeError: logger.info(api_result) raise with suppress(KeyError): api_result['result']['request_headers'] = CaseInsensitiveDict(api_result['result']['request_headers']) api_result['result']['response_headers'] = CaseInsensitiveDict(api_result['result']['response_headers']) if self.large_object_handler is not None and api_result['result']['content']: content_format = api_result['result']['format'] if content_format in ['clob', 'blob']: api_result['result']['content'], api_result['result']['format'] = self.large_object_handler(callback_url=api_result['result']['content'], format=content_format) elif content_format == 'binary': base64_payload = api_result['result']['content'] if isinstance(base64_payload, bytes): base64_payload = base64_payload.decode('utf-8') api_result['result']['content'] = BytesIO(b64decode(base64_payload)) return FrozenDict(api_result) def _is_api_error(self, api_result: Dict) -> bool: if self.scrape_config.method == 'HEAD': if 'X-Reject-Reason' in self.response.headers: return True return False if api_result is None: return True return 'error_id' in api_result def upstream_result_into_response(self, _class=Response) -> Optional[Response]: if _class != Response: raise RuntimeError('only Response from requests package is supported at the moment') if self.result is None: return None if self.response.status_code != 200: return None response = Response() response.status_code = self.scrape_result['status_code'] response.reason = self.scrape_result['reason'] if self.scrape_result['content']: if isinstance(self.scrape_result['content'], BytesIO): response._content = self.scrape_result['content'].getvalue() elif isinstance(self.scrape_result['content'], bytes): response._content = self.scrape_result['content'] elif isinstance(self.scrape_result['content'], str): response._content = self.scrape_result['content'].encode('utf-8') else: response._content = None response.headers.update(self.scrape_result['response_headers']) response.url = self.scrape_result['url'] response.request = Request( method=self.config['method'], url=self.config['url'], headers=self.scrape_result['request_headers'], data=self.config['body'] if self.config['body'] else None ) if 'set-cookie' in response.headers: for raw_cookie in response.headers['set-cookie']: for name, cookie in SimpleCookie(raw_cookie).items(): expires = cookie.get('expires') if expires == '': expires = None if expires: try: expires = parse(expires).timestamp() except ValueError: expires = None if type(expires) == str: if '.' in expires: expires = float(expires) else: expires = int(expires) response.cookies.set_cookie(Cookie( version=cookie.get('version') if cookie.get('version') else None, name=name, value=cookie.value, path=cookie.get('path', ''), expires=expires, comment=cookie.get('comment'), domain=cookie.get('domain', ''), secure=cookie.get('secure'), port=None, port_specified=False, domain_specified=cookie.get('domain') is not None and cookie.get('domain') != '', domain_initial_dot=bool(cookie.get('domain').startswith('.')) if cookie.get('domain') is not None else False, path_specified=cookie.get('path') != '' and cookie.get('path') is not None, discard=False, comment_url=None, rest={ 'httponly': cookie.get('httponly'), 'samesite': cookie.get('samesite'), 'max-age': cookie.get('max-age') } )) return response def sink(self, path: Optional[str] = None, name: Optional[str] = None, file: Optional[Union[TextIO, BytesIO]] = None, content: Optional[Union[str, bytes]] = None): file_content = content or self.scrape_result['content'] file_path = None file_extension = None if name: name_parts = name.split('.') if len(name_parts) > 1: file_extension = name_parts[-1] if not file: if file_extension is None: try: mime_type = self.scrape_result['response_headers']['content-type'] except KeyError: mime_type = 'application/octet-stream' if ';' in mime_type: mime_type = mime_type.split(';')[0] file_extension = '.' + mime_type.split('/')[1] if not name: name = self.config['url'].split('/')[-1] if name.find(file_extension) == -1: name += file_extension file_path = path + '/' + name if path is not None else name if file_path == file_extension: url = re.sub(r'(https|http)?://', '', self.config['url']).replace('/', '-') if url[-1] == '-': url = url[:-1] url += file_extension file_path = url file = open(file_path, 'wb') if isinstance(file_content, str): file_content = BytesIO(file_content.encode('utf-8')) elif isinstance(file_content, bytes): file_content = BytesIO(file_content) file_content.seek(0) with file as f: shutil.copyfileobj(file_content, f, length=131072) logger.info('file %s created' % file_path) def raise_for_result(self, raise_on_upstream_error=True, error_class=ApiHttpClientError): super().raise_for_result(raise_on_upstream_error=raise_on_upstream_error, error_class=error_class) if self.result['result']['status'] == 'DONE' and self.scrape_success is False: error = ErrorFactory.create(api_response=self) if error: if isinstance(error, UpstreamHttpError): if raise_on_upstream_error is True: raise error else: raise errorAncestors
Class variables
var large_object_handler : Callablevar scrape_config : ScrapeConfig
Instance variables
prop config : Dict | None-
Expand source code
@property def config(self) -> Optional[Dict]: if self.scrape_result is None: return None return self.result['config'] prop content : str-
Expand source code
@property def content(self) -> str: if self.scrape_result is None: return '' return self.scrape_result['content'] prop context : Dict | None-
Expand source code
@property def context(self) -> Optional[Dict]: if self.scrape_result is None: return None return self.result['context'] prop error : Dict | None-
Expand source code
@property def error(self) -> Optional[Dict]: if self.scrape_result is None: return None if self.scrape_success is False: return self.scrape_result.get('error') prop scrape_result : Dict | None-
Expand source code
@property def scrape_result(self) -> Optional[Dict]: return self.result.get('result', None) prop scrape_success : bool-
Expand source code
@property def scrape_success(self) -> bool: scrape_result = self.scrape_result if not scrape_result: return False return self.scrape_result['success'] var selector : Selector-
Expand source code
@cached_property def selector(self) -> 'Selector': if self.scrape_result['format'] != 'text': raise ContentError("Unable to cast into beautiful soup, the format of data is binary - must be text content") try: from parsel import Selector return Selector(text=self.content) except ImportError as e: logger.error('You must install parsel or scrapy package to enable this feature') raise e var soup : BeautifulSoup-
Expand source code
@cached_property def soup(self) -> 'BeautifulSoup': if self.scrape_result['format'] != 'text': raise ContentError("Unable to cast into beautiful soup, the format of data is binary - must be text content") try: from bs4 import BeautifulSoup soup = BeautifulSoup(self.content, "lxml") return soup except ImportError as e: logger.error('You must install scrapfly[parser] to enable this feature') prop success : bool-
Expand source code
@property def success(self) -> bool: """ Success means Scrapfly api reply correctly to the call, but the scrape can be unsuccessful if the upstream reply with error status code """ return 200 <= self.response.status_code <= 299Success means Scrapfly api reply correctly to the call, but the scrape can be unsuccessful if the upstream reply with error status code
prop upstream_status_code : int | None-
Expand source code
@property def upstream_status_code(self) -> Optional[int]: if self.scrape_result is None: return None if 'status_code' in self.scrape_result: return self.scrape_result['status_code'] return None
Methods
def handle_api_result(self, api_result: Dict) ‑> FrozenDict | None-
Expand source code
def handle_api_result(self, api_result: Dict) -> Optional[FrozenDict]: if self._is_api_error(api_result=api_result) is True: return FrozenDict(api_result) try: if isinstance(api_result['config']['headers'], list): api_result['config']['headers'] = {} except TypeError: logger.info(api_result) raise with suppress(KeyError): api_result['result']['request_headers'] = CaseInsensitiveDict(api_result['result']['request_headers']) api_result['result']['response_headers'] = CaseInsensitiveDict(api_result['result']['response_headers']) if self.large_object_handler is not None and api_result['result']['content']: content_format = api_result['result']['format'] if content_format in ['clob', 'blob']: api_result['result']['content'], api_result['result']['format'] = self.large_object_handler(callback_url=api_result['result']['content'], format=content_format) elif content_format == 'binary': base64_payload = api_result['result']['content'] if isinstance(base64_payload, bytes): base64_payload = base64_payload.decode('utf-8') api_result['result']['content'] = BytesIO(b64decode(base64_payload)) return FrozenDict(api_result) def raise_for_result(self,
raise_on_upstream_error=True,
error_class=scrapfly.errors.ApiHttpClientError)-
Expand source code
def raise_for_result(self, raise_on_upstream_error=True, error_class=ApiHttpClientError): super().raise_for_result(raise_on_upstream_error=raise_on_upstream_error, error_class=error_class) if self.result['result']['status'] == 'DONE' and self.scrape_success is False: error = ErrorFactory.create(api_response=self) if error: if isinstance(error, UpstreamHttpError): if raise_on_upstream_error is True: raise error else: raise error def sink(self,
path: str | None = None,
name: str | None = None,
file:| _io.BytesIO | None = None,
content: str | bytes | None = None)-
Expand source code
def sink(self, path: Optional[str] = None, name: Optional[str] = None, file: Optional[Union[TextIO, BytesIO]] = None, content: Optional[Union[str, bytes]] = None): file_content = content or self.scrape_result['content'] file_path = None file_extension = None if name: name_parts = name.split('.') if len(name_parts) > 1: file_extension = name_parts[-1] if not file: if file_extension is None: try: mime_type = self.scrape_result['response_headers']['content-type'] except KeyError: mime_type = 'application/octet-stream' if ';' in mime_type: mime_type = mime_type.split(';')[0] file_extension = '.' + mime_type.split('/')[1] if not name: name = self.config['url'].split('/')[-1] if name.find(file_extension) == -1: name += file_extension file_path = path + '/' + name if path is not None else name if file_path == file_extension: url = re.sub(r'(https|http)?://', '', self.config['url']).replace('/', '-') if url[-1] == '-': url = url[:-1] url += file_extension file_path = url file = open(file_path, 'wb') if isinstance(file_content, str): file_content = BytesIO(file_content.encode('utf-8')) elif isinstance(file_content, bytes): file_content = BytesIO(file_content) file_content.seek(0) with file as f: shutil.copyfileobj(file_content, f, length=131072) logger.info('file %s created' % file_path) def upstream_result_into_response(self) ‑> requests.models.Response | None-
Expand source code
def upstream_result_into_response(self, _class=Response) -> Optional[Response]: if _class != Response: raise RuntimeError('only Response from requests package is supported at the moment') if self.result is None: return None if self.response.status_code != 200: return None response = Response() response.status_code = self.scrape_result['status_code'] response.reason = self.scrape_result['reason'] if self.scrape_result['content']: if isinstance(self.scrape_result['content'], BytesIO): response._content = self.scrape_result['content'].getvalue() elif isinstance(self.scrape_result['content'], bytes): response._content = self.scrape_result['content'] elif isinstance(self.scrape_result['content'], str): response._content = self.scrape_result['content'].encode('utf-8') else: response._content = None response.headers.update(self.scrape_result['response_headers']) response.url = self.scrape_result['url'] response.request = Request( method=self.config['method'], url=self.config['url'], headers=self.scrape_result['request_headers'], data=self.config['body'] if self.config['body'] else None ) if 'set-cookie' in response.headers: for raw_cookie in response.headers['set-cookie']: for name, cookie in SimpleCookie(raw_cookie).items(): expires = cookie.get('expires') if expires == '': expires = None if expires: try: expires = parse(expires).timestamp() except ValueError: expires = None if type(expires) == str: if '.' in expires: expires = float(expires) else: expires = int(expires) response.cookies.set_cookie(Cookie( version=cookie.get('version') if cookie.get('version') else None, name=name, value=cookie.value, path=cookie.get('path', ''), expires=expires, comment=cookie.get('comment'), domain=cookie.get('domain', ''), secure=cookie.get('secure'), port=None, port_specified=False, domain_specified=cookie.get('domain') is not None and cookie.get('domain') != '', domain_initial_dot=bool(cookie.get('domain').startswith('.')) if cookie.get('domain') is not None else False, path_specified=cookie.get('path') != '' and cookie.get('path') is not None, discard=False, comment_url=None, rest={ 'httponly': cookie.get('httponly'), 'samesite': cookie.get('samesite'), 'max-age': cookie.get('max-age') } )) return response
Inherited members
class ScrapeConfig (url: str,
retry: bool = True,
method: str = 'GET',
country: str | None = None,
render_js: bool = False,
cache: bool = False,
cache_clear: bool = False,
ssl: bool = False,
dns: bool = False,
asp: bool | scrapfly.scrape_config._Unset = <unset>,
debug: bool = False,
raise_on_upstream_error: bool = True,
cache_ttl: int | None = None,
proxy_pool: str | None = None,
session: str | None = None,
tags: List[str] | Set[str] | None = None,
format: Format | None = None,
format_options: List[FormatOption] | None = None,
extraction_template: str | None = None,
extraction_ephemeral_template: Dict | None = None,
extraction_prompt: str | None = None,
extraction_model: str | None = None,
correlation_id: str | None = None,
cookies: requests.structures.CaseInsensitiveDict | None = None,
body: str | None = None,
data: Dict | None = None,
headers: requests.structures.CaseInsensitiveDict | Dict[str, str] | None = None,
js: str = None,
rendering_wait: int = None,
rendering_stage: Literal['complete', 'domcontentloaded'] = 'complete',
wait_for_selector: str | None = None,
screenshots: Dict | None = None,
screenshot_flags: List[ScreenshotFlag] | None = None,
session_sticky_proxy: bool = True,
webhook: str | None = None,
timeout: int | None = None,
js_scenario: List | None = None,
extract: Dict | None = None,
os: str | None = None,
lang: List[str] | None = None,
auto_scroll: bool | None = None,
cost_budget: int | None = None,
browser_brand: str | None = None,
geolocation: str | None = None,
proxified_response: bool | None = None,
unblocker: bool | scrapfly.scrape_config._Unset = <unset>)-
Expand source code
class ScrapeConfig(BaseApiConfig): PUBLIC_DATACENTER_POOL = 'public_datacenter_pool' PUBLIC_RESIDENTIAL_POOL = 'public_residential_pool' PUBLIC_TOR_POOL = 'public_tor_pool' url: str retry: bool = True method: str = 'GET' country: Optional[str] = None render_js: bool = False cache: bool = False cache_clear:bool = False ssl:bool = False dns:bool = False asp:bool = False # deprecated alias of `unblocker`, permanently supported # NOTE: `unblocker` is deliberately absent from this block. It is a # property defined below; a class attribute or annotation here would # shadow the descriptor and break `cfg.unblocker = True`. debug: bool = False raise_on_upstream_error:bool = True cache_ttl:Optional[int] = None proxy_pool:Optional[str] = None session: Optional[str] = None tags: Optional[List[str]] = None format: Optional[Format] = None, # raw(unchanged) format_options: Optional[List[FormatOption]] extraction_template: Optional[str] = None # a saved template name extraction_ephemeral_template: Optional[Dict] # ephemeraly declared json template extraction_prompt: Optional[str] = None extraction_model: Optional[str] = None correlation_id: Optional[str] = None cookies: Optional[CaseInsensitiveDict] = None body: Optional[str] = None data: Optional[Dict] = None headers: Optional[CaseInsensitiveDict] = None js: str = None rendering_wait: int = None rendering_stage: Literal["complete", "domcontentloaded"] = "complete" wait_for_selector: Optional[str] = None session_sticky_proxy:bool = True screenshots:Optional[Dict]=None screenshot_flags: Optional[List[ScreenshotFlag]] = None, webhook:Optional[str]=None timeout:Optional[int]=None # in milliseconds js_scenario: Dict = None extract: Dict = None lang:Optional[List[str]] = None os:Optional[str] = None auto_scroll:Optional[bool] = None cost_budget:Optional[int] = None browser_brand:Optional[str] = None geolocation:Optional[str] = None proxified_response:Optional[bool] = None def __init__( self, url: str, retry: bool = True, method: str = 'GET', country: Optional[str] = None, render_js: bool = False, cache: bool = False, cache_clear:bool = False, ssl:bool = False, dns:bool = False, asp:Union[bool, _Unset] = _UNSET, # deprecated alias of `unblocker`, which is declared last debug: bool = False, raise_on_upstream_error:bool = True, cache_ttl:Optional[int] = None, proxy_pool:Optional[str] = None, session: Optional[str] = None, tags: Optional[Union[List[str], Set[str]]] = None, format: Optional[Format] = None, # raw(unchanged) format_options: Optional[List[FormatOption]] = None, # raw(unchanged) extraction_template: Optional[str] = None, # a saved template name extraction_ephemeral_template: Optional[Dict] = None, # ephemeraly declared json template extraction_prompt: Optional[str] = None, extraction_model: Optional[str] = None, correlation_id: Optional[str] = None, cookies: Optional[CaseInsensitiveDict] = None, body: Optional[str] = None, data: Optional[Dict] = None, headers: Optional[Union[CaseInsensitiveDict, Dict[str, str]]] = None, js: str = None, rendering_wait: int = None, rendering_stage: Literal["complete", "domcontentloaded"] = "complete", wait_for_selector: Optional[str] = None, screenshots:Optional[Dict]=None, screenshot_flags: Optional[List[ScreenshotFlag]] = None, session_sticky_proxy:bool = True, webhook:Optional[str] = None, timeout:Optional[int] = None, # in milliseconds js_scenario:Optional[List] = None, extract:Optional[Dict] = None, os:Optional[str] = None, lang:Optional[List[str]] = None, auto_scroll:Optional[bool] = None, cost_budget:Optional[int] = None, browser_brand:Optional[str] = None, geolocation:Optional[str] = None, proxified_response:Optional[bool] = None, # Appended at the very END of the signature on purpose: this parameter # list is positional-capable, so inserting `unblocker` next to its # alias `asp` would silently shift every positional argument in # existing customer code. unblocker:Union[bool, _Unset] = _UNSET ): assert(type(url) is str) if isinstance(tags, List): tags = set(tags) cookies = cookies or {} headers = headers or {} self.cookies = CaseInsensitiveDict(cookies) self.headers = CaseInsensitiveDict(headers) self.url = url self.retry = retry self.method = method self.country = country self.session_sticky_proxy = session_sticky_proxy self.render_js = render_js self.cache = cache self.cache_clear = cache_clear # Both names normalize into the single stored attribute `self.asp`, # which is also the key emitted on the wire. self.asp = _resolve_unblocker(asp, unblocker) self.webhook = webhook self.session = session self.debug = debug self.cache_ttl = cache_ttl self.proxy_pool = proxy_pool self.tags = tags or set() self.format = format self.format_options = format_options self.extraction_template = extraction_template self.extraction_ephemeral_template = extraction_ephemeral_template self.extraction_prompt = extraction_prompt self.extraction_model = extraction_model self.correlation_id = correlation_id self.wait_for_selector = wait_for_selector self.body = body self.data = data self.js = js self.rendering_wait = rendering_wait self.rendering_stage = rendering_stage self.raise_on_upstream_error = raise_on_upstream_error self.screenshots = screenshots self.screenshot_flags = screenshot_flags self.key = None self.dns = dns self.ssl = ssl self.js_scenario = js_scenario self.timeout = timeout self.extract = extract self.lang = lang self.os = os self.auto_scroll = auto_scroll self.cost_budget = cost_budget self.browser_brand = browser_brand self.geolocation = geolocation self.proxified_response = proxified_response if cookies: _cookies = [] for name, value in cookies.items(): _cookies.append(name + '=' + value) if 'cookie' in self.headers: if self.headers['cookie'][-1] != ';': self.headers['cookie'] += ';' else: self.headers['cookie'] = '' self.headers['cookie'] += '; '.join(_cookies) if self.body and self.data: raise ScrapeConfigError('You cannot pass both parameters body and data. You must choose') if method in ['POST', 'PUT', 'PATCH']: if self.body is None and self.data is not None: if 'content-type' not in self.headers: self.headers['content-type'] = 'application/x-www-form-urlencoded' self.body = urlencode(data) else: if self.headers['content-type'].find('application/json') != -1: self.body = json.dumps(data) elif self.headers['content-type'].find('application/x-www-form-urlencoded') != -1: self.body = urlencode(data) else: raise ScrapeConfigError('Content-Type "%s" not supported, use body parameter to pass pre encoded body according to your content type' % self.headers['content-type']) elif self.body is None and self.data is None: self.headers['content-type'] = 'text/plain' @property def unblocker(self) -> bool: """Anti-bot bypass toggle, the current name for what used to be `asp`. Deliberately a property rather than a second stored attribute: it stays out of `__dict__`, so serializers that iterate the instance dict do not sprout a duplicate key, while `config.unblocker = True` after construction still works. """ return self.asp @unblocker.setter def unblocker(self, value: bool): self.asp = value def to_api_params(self, key:str) -> Dict: params = { 'key': self.key or key, 'url': self.url } if self.country is not None: params['country'] = self.country for name, value in self.headers.items(): params['headers[%s]' % name] = value if self.webhook is not None: params['webhook_name'] = self.webhook if self.timeout is not None: params['timeout'] = self.timeout if self.extract is not None: params['extract'] = base64.urlsafe_b64encode(json.dumps(self.extract).encode('utf-8')).decode('utf-8') if self.cost_budget is not None: params['cost_budget'] = self.cost_budget if self.proxified_response is not None: params['proxified_response'] = self._bool_to_http(self.proxified_response) if self.render_js is True: params['render_js'] = self._bool_to_http(self.render_js) if self.wait_for_selector is not None: params['wait_for_selector'] = self.wait_for_selector if self.js: params['js'] = base64.urlsafe_b64encode(self.js.encode('utf-8')).decode('utf-8') if self.js_scenario: params['js_scenario'] = base64.urlsafe_b64encode(json.dumps(self.js_scenario).encode('utf-8')).decode('utf-8') if self.rendering_wait: params['rendering_wait'] = self.rendering_wait if self.rendering_stage: params['rendering_stage'] = self.rendering_stage if self.screenshots is not None: for name, element in self.screenshots.items(): params['screenshots[%s]' % name] = element if self.screenshot_flags is not None: self.screenshot_flags = [ScreenshotFlag(flag) for flag in self.screenshot_flags] params["screenshot_flags"] = ",".join(flag.value for flag in self.screenshot_flags) else: if self.screenshot_flags is not None: logging.warning('Params "screenshot_flags" is ignored. Works only if screenshots is enabled') if self.auto_scroll is True: params['auto_scroll'] = self._bool_to_http(self.auto_scroll) else: if self.wait_for_selector is not None: logging.warning('Params "wait_for_selector" is ignored. Works only if render_js is enabled') if self.screenshots: logging.warning('Params "screenshots" is ignored. Works only if render_js is enabled') if self.js_scenario: logging.warning('Params "js_scenario" is ignored. Works only if render_js is enabled') if self.js: logging.warning('Params "js" is ignored. Works only if render_js is enabled') if self.rendering_wait: logging.warning('Params "rendering_wait" is ignored. Works only if render_js is enabled') # The emitted key stays `asp`. Published SDK versions are immutable and # upgraded per installation, so emitting `unblocker` against an API # deployment that has not learned it yet would silently drop a paid # feature: the scrape succeeds, is billed, and returns a blocked page. if self.asp is True: params['asp'] = self._bool_to_http(self.asp) if self.retry is False: params['retry'] = self._bool_to_http(self.retry) if self.cache is True: params['cache'] = self._bool_to_http(self.cache) if self.cache_clear is True: params['cache_clear'] = self._bool_to_http(self.cache_clear) if self.cache_ttl is not None: params['cache_ttl'] = self.cache_ttl else: if self.cache_clear is True: logging.warning('Params "cache_clear" is ignored. Works only if cache is enabled') if self.cache_ttl is not None: logging.warning('Params "cache_ttl" is ignored. Works only if cache is enabled') if self.dns is True: params['dns'] = self._bool_to_http(self.dns) if self.ssl is True: params['ssl'] = self._bool_to_http(self.ssl) if self.tags: params['tags'] = ','.join(self.tags) if self.format: params['format'] = Format(self.format).value if self.format_options: params['format'] += ':' + ','.join(FormatOption(option).value for option in self.format_options) if self.extraction_template and self.extraction_ephemeral_template: raise ScrapeConfigError('You cannot pass both parameters extraction_template and extraction_ephemeral_template. You must choose') if self.extraction_template: params['extraction_template'] = 'persistent:' + self.extraction_template if self.extraction_ephemeral_template: self.extraction_ephemeral_template = json.dumps(self.extraction_ephemeral_template) params['extraction_template'] = 'ephemeral:' + urlsafe_b64encode(self.extraction_ephemeral_template.encode('utf-8')).decode('utf-8') if self.extraction_prompt: params['extraction_prompt'] = quote_plus(self.extraction_prompt) if self.extraction_model: params['extraction_model'] = self.extraction_model if self.correlation_id: params['correlation_id'] = self.correlation_id if self.session: params['session'] = self.session if self.session_sticky_proxy is not None: params['session_sticky_proxy'] = self._bool_to_http(self.session_sticky_proxy) else: if self.session_sticky_proxy: logging.warning('Params "session_sticky_proxy" is ignored. Works only if session is enabled') if self.debug is True: params['debug'] = self._bool_to_http(self.debug) if self.proxy_pool is not None: params['proxy_pool'] = self.proxy_pool if self.lang is not None: params['lang'] = ','.join(self.lang) if self.os is not None: params['os'] = self.os if self.browser_brand is not None: params['browser_brand'] = self.browser_brand if self.geolocation is not None: params['geolocation'] = self.geolocation return params @staticmethod def from_exported_config(config:str) -> 'ScrapeConfig': try: from msgpack import loads as msgpack_loads except ImportError as e: print('You must install msgpack package - run: pip install "scrapfly-sdk[speedups]" or pip install msgpack') raise data = msgpack_loads(base64.b64decode(config)) headers = {} for name, value in data['headers'].items(): if isinstance(value, Iterable): headers[name] = '; '.join(value) else: headers[name] = value return ScrapeConfig( url=data['url'], retry=data['retry'], headers=headers, session=data['session'], session_sticky_proxy=data['session_sticky_proxy'], cache=data['cache'], cache_ttl=data['cache_ttl'], cache_clear=data['cache_clear'], render_js=data['render_js'], method=data['method'], # .get(), not a subscript: an export produced under either name # must load, and neither key is guaranteed present. asp=data.get('asp', _UNSET), unblocker=data.get('unblocker', _UNSET), body=data['body'], ssl=data['ssl'], dns=data['dns'], country=data['country'], debug=data['debug'], correlation_id=data['correlation_id'], tags=data['tags'], format=data['format'], js=data['js'], rendering_wait=data['rendering_wait'], screenshots=data['screenshots'] or {}, screenshot_flags=data['screenshot_flags'], proxy_pool=data['proxy_pool'], auto_scroll=data['auto_scroll'], cost_budget=data['cost_budget'] ) def to_dict(self) -> Dict: """ Export the ScrapeConfig instance to a plain dictionary. Useful for JSON-serialization or other external storage. """ return { 'url': self.url, 'retry': self.retry, 'method': self.method, 'country': self.country, 'render_js': self.render_js, 'cache': self.cache, 'cache_clear': self.cache_clear, 'ssl': self.ssl, 'dns': self.dns, 'asp': self.asp, 'debug': self.debug, 'raise_on_upstream_error': self.raise_on_upstream_error, 'cache_ttl': self.cache_ttl, 'proxy_pool': self.proxy_pool, 'session': self.session, 'tags': list(self.tags), 'format': Format(self.format).value if self.format else None, 'format_options': [FormatOption(option).value for option in self.format_options] if self.format_options else None, 'extraction_template': self.extraction_template, 'extraction_ephemeral_template': self.extraction_ephemeral_template, 'extraction_prompt': self.extraction_prompt, 'extraction_model': self.extraction_model, 'correlation_id': self.correlation_id, 'cookies': CaseInsensitiveDict(self.cookies), 'body': self.body, 'data': None if self.body else self.data, 'headers': CaseInsensitiveDict(self.headers), 'js': self.js, 'rendering_wait': self.rendering_wait, 'wait_for_selector': self.wait_for_selector, 'session_sticky_proxy': self.session_sticky_proxy, 'screenshots': self.screenshots, 'screenshot_flags': [ScreenshotFlag(flag).value for flag in self.screenshot_flags] if self.screenshot_flags else None, 'webhook': self.webhook, 'timeout': self.timeout, 'js_scenario': self.js_scenario, 'extract': self.extract, 'lang': self.lang, 'os': self.os, 'auto_scroll': self.auto_scroll, 'cost_budget': self.cost_budget, 'browser_brand': self.browser_brand, } @staticmethod def from_dict(scrape_config_dict: Dict) -> 'ScrapeConfig': """Create a ScrapeConfig instance from a dictionary.""" url = scrape_config_dict.get('url', None) retry = scrape_config_dict.get('retry', False) method = scrape_config_dict.get('method', 'GET') country = scrape_config_dict.get('country', None) render_js = scrape_config_dict.get('render_js', False) cache = scrape_config_dict.get('cache', False) cache_clear = scrape_config_dict.get('cache_clear', False) ssl = scrape_config_dict.get('ssl', False) dns = scrape_config_dict.get('dns', False) # Sentinel defaults, not False: an explicit False under either key must # stay distinguishable from an absent key so precedence works. asp = scrape_config_dict.get('asp', _UNSET) unblocker = scrape_config_dict.get('unblocker', _UNSET) debug = scrape_config_dict.get('debug', False) raise_on_upstream_error = scrape_config_dict.get('raise_on_upstream_error', True) cache_ttl = scrape_config_dict.get('cache_ttl', None) proxy_pool = scrape_config_dict.get('proxy_pool', None) session = scrape_config_dict.get('session', None) tags = scrape_config_dict.get('tags', []) format = scrape_config_dict.get('format', None) format = Format(format) if format else None format_options = scrape_config_dict.get('format_options', None) format_options = [FormatOption(option) for option in format_options] if format_options else None extraction_template = scrape_config_dict.get('extraction_template', None) extraction_ephemeral_template = scrape_config_dict.get('extraction_ephemeral_template', None) extraction_prompt = scrape_config_dict.get('extraction_prompt', None) extraction_model = scrape_config_dict.get('extraction_model', None) correlation_id = scrape_config_dict.get('correlation_id', None) cookies = scrape_config_dict.get('cookies', {}) body = scrape_config_dict.get('body', None) data = scrape_config_dict.get('data', None) headers = scrape_config_dict.get('headers', {}) js = scrape_config_dict.get('js', None) rendering_wait = scrape_config_dict.get('rendering_wait', None) wait_for_selector = scrape_config_dict.get('wait_for_selector', None) screenshots = scrape_config_dict.get('screenshots', []) screenshot_flags = scrape_config_dict.get('screenshot_flags', []) screenshot_flags = [ScreenshotFlag(flag) for flag in screenshot_flags] if screenshot_flags else None session_sticky_proxy = scrape_config_dict.get('session_sticky_proxy', True) webhook = scrape_config_dict.get('webhook', None) timeout = scrape_config_dict.get('timeout', None) js_scenario = scrape_config_dict.get('js_scenario', None) extract = scrape_config_dict.get('extract', None) os = scrape_config_dict.get('os', None) lang = scrape_config_dict.get('lang', None) auto_scroll = scrape_config_dict.get('auto_scroll', None) cost_budget = scrape_config_dict.get('cost_budget', None) browser_brand = scrape_config_dict.get('browser_brand', None) return ScrapeConfig( url=url, retry=retry, method=method, country=country, render_js=render_js, cache=cache, cache_clear=cache_clear, ssl=ssl, dns=dns, asp=asp, debug=debug, raise_on_upstream_error=raise_on_upstream_error, cache_ttl=cache_ttl, proxy_pool=proxy_pool, session=session, tags=tags, format=format, format_options=format_options, extraction_template=extraction_template, extraction_ephemeral_template=extraction_ephemeral_template, extraction_prompt=extraction_prompt, extraction_model=extraction_model, correlation_id=correlation_id, cookies=cookies, body=body, data=data, headers=headers, js=js, rendering_wait=rendering_wait, wait_for_selector=wait_for_selector, screenshots=screenshots, screenshot_flags=screenshot_flags, session_sticky_proxy=session_sticky_proxy, webhook=webhook, timeout=timeout, js_scenario=js_scenario, extract=extract, os=os, lang=lang, auto_scroll=auto_scroll, cost_budget=cost_budget, browser_brand=browser_brand, unblocker=unblocker, )Ancestors
Class variables
var PUBLIC_DATACENTER_POOLvar PUBLIC_RESIDENTIAL_POOLvar PUBLIC_TOR_POOLvar asp : boolvar auto_scroll : bool | Nonevar body : str | Nonevar browser_brand : str | Nonevar cache : boolvar cache_clear : boolvar cache_ttl : int | Nonevar correlation_id : str | Nonevar cost_budget : int | Nonevar country : str | Nonevar data : Dict | Nonevar debug : boolvar dns : boolvar extract : Dictvar extraction_ephemeral_template : Dict | Nonevar extraction_model : str | Nonevar extraction_prompt : str | Nonevar extraction_template : str | Nonevar format : Format | Nonevar format_options : List[FormatOption] | Nonevar geolocation : str | Nonevar headers : requests.structures.CaseInsensitiveDict | Nonevar js : strvar js_scenario : Dictvar lang : List[str] | Nonevar method : strvar os : str | Nonevar proxified_response : bool | Nonevar proxy_pool : str | Nonevar raise_on_upstream_error : boolvar render_js : boolvar rendering_stage : Literal['complete', 'domcontentloaded']var rendering_wait : intvar retry : boolvar screenshot_flags : List[ScreenshotFlag] | Nonevar screenshots : Dict | Nonevar session : str | Nonevar session_sticky_proxy : boolvar ssl : boolvar timeout : int | Nonevar url : strvar wait_for_selector : str | Nonevar webhook : str | None
Static methods
def from_dict(scrape_config_dict: Dict) ‑> ScrapeConfig-
Expand source code
@staticmethod def from_dict(scrape_config_dict: Dict) -> 'ScrapeConfig': """Create a ScrapeConfig instance from a dictionary.""" url = scrape_config_dict.get('url', None) retry = scrape_config_dict.get('retry', False) method = scrape_config_dict.get('method', 'GET') country = scrape_config_dict.get('country', None) render_js = scrape_config_dict.get('render_js', False) cache = scrape_config_dict.get('cache', False) cache_clear = scrape_config_dict.get('cache_clear', False) ssl = scrape_config_dict.get('ssl', False) dns = scrape_config_dict.get('dns', False) # Sentinel defaults, not False: an explicit False under either key must # stay distinguishable from an absent key so precedence works. asp = scrape_config_dict.get('asp', _UNSET) unblocker = scrape_config_dict.get('unblocker', _UNSET) debug = scrape_config_dict.get('debug', False) raise_on_upstream_error = scrape_config_dict.get('raise_on_upstream_error', True) cache_ttl = scrape_config_dict.get('cache_ttl', None) proxy_pool = scrape_config_dict.get('proxy_pool', None) session = scrape_config_dict.get('session', None) tags = scrape_config_dict.get('tags', []) format = scrape_config_dict.get('format', None) format = Format(format) if format else None format_options = scrape_config_dict.get('format_options', None) format_options = [FormatOption(option) for option in format_options] if format_options else None extraction_template = scrape_config_dict.get('extraction_template', None) extraction_ephemeral_template = scrape_config_dict.get('extraction_ephemeral_template', None) extraction_prompt = scrape_config_dict.get('extraction_prompt', None) extraction_model = scrape_config_dict.get('extraction_model', None) correlation_id = scrape_config_dict.get('correlation_id', None) cookies = scrape_config_dict.get('cookies', {}) body = scrape_config_dict.get('body', None) data = scrape_config_dict.get('data', None) headers = scrape_config_dict.get('headers', {}) js = scrape_config_dict.get('js', None) rendering_wait = scrape_config_dict.get('rendering_wait', None) wait_for_selector = scrape_config_dict.get('wait_for_selector', None) screenshots = scrape_config_dict.get('screenshots', []) screenshot_flags = scrape_config_dict.get('screenshot_flags', []) screenshot_flags = [ScreenshotFlag(flag) for flag in screenshot_flags] if screenshot_flags else None session_sticky_proxy = scrape_config_dict.get('session_sticky_proxy', True) webhook = scrape_config_dict.get('webhook', None) timeout = scrape_config_dict.get('timeout', None) js_scenario = scrape_config_dict.get('js_scenario', None) extract = scrape_config_dict.get('extract', None) os = scrape_config_dict.get('os', None) lang = scrape_config_dict.get('lang', None) auto_scroll = scrape_config_dict.get('auto_scroll', None) cost_budget = scrape_config_dict.get('cost_budget', None) browser_brand = scrape_config_dict.get('browser_brand', None) return ScrapeConfig( url=url, retry=retry, method=method, country=country, render_js=render_js, cache=cache, cache_clear=cache_clear, ssl=ssl, dns=dns, asp=asp, debug=debug, raise_on_upstream_error=raise_on_upstream_error, cache_ttl=cache_ttl, proxy_pool=proxy_pool, session=session, tags=tags, format=format, format_options=format_options, extraction_template=extraction_template, extraction_ephemeral_template=extraction_ephemeral_template, extraction_prompt=extraction_prompt, extraction_model=extraction_model, correlation_id=correlation_id, cookies=cookies, body=body, data=data, headers=headers, js=js, rendering_wait=rendering_wait, wait_for_selector=wait_for_selector, screenshots=screenshots, screenshot_flags=screenshot_flags, session_sticky_proxy=session_sticky_proxy, webhook=webhook, timeout=timeout, js_scenario=js_scenario, extract=extract, os=os, lang=lang, auto_scroll=auto_scroll, cost_budget=cost_budget, browser_brand=browser_brand, unblocker=unblocker, )Create a ScrapeConfig instance from a dictionary.
def from_exported_config(config: str) ‑> ScrapeConfig-
Expand source code
@staticmethod def from_exported_config(config:str) -> 'ScrapeConfig': try: from msgpack import loads as msgpack_loads except ImportError as e: print('You must install msgpack package - run: pip install "scrapfly-sdk[speedups]" or pip install msgpack') raise data = msgpack_loads(base64.b64decode(config)) headers = {} for name, value in data['headers'].items(): if isinstance(value, Iterable): headers[name] = '; '.join(value) else: headers[name] = value return ScrapeConfig( url=data['url'], retry=data['retry'], headers=headers, session=data['session'], session_sticky_proxy=data['session_sticky_proxy'], cache=data['cache'], cache_ttl=data['cache_ttl'], cache_clear=data['cache_clear'], render_js=data['render_js'], method=data['method'], # .get(), not a subscript: an export produced under either name # must load, and neither key is guaranteed present. asp=data.get('asp', _UNSET), unblocker=data.get('unblocker', _UNSET), body=data['body'], ssl=data['ssl'], dns=data['dns'], country=data['country'], debug=data['debug'], correlation_id=data['correlation_id'], tags=data['tags'], format=data['format'], js=data['js'], rendering_wait=data['rendering_wait'], screenshots=data['screenshots'] or {}, screenshot_flags=data['screenshot_flags'], proxy_pool=data['proxy_pool'], auto_scroll=data['auto_scroll'], cost_budget=data['cost_budget'] )
Instance variables
prop unblocker : bool-
Expand source code
@property def unblocker(self) -> bool: """Anti-bot bypass toggle, the current name for what used to be `asp`. Deliberately a property rather than a second stored attribute: it stays out of `__dict__`, so serializers that iterate the instance dict do not sprout a duplicate key, while `config.unblocker = True` after construction still works. """ return self.aspAnti-bot bypass toggle, the current name for what used to be
asp.Deliberately a property rather than a second stored attribute: it stays out of
__dict__, so serializers that iterate the instance dict do not sprout a duplicate key, whileconfig.unblocker = Trueafter construction still works.
Methods
def to_api_params(self, key: str) ‑> Dict-
Expand source code
def to_api_params(self, key:str) -> Dict: params = { 'key': self.key or key, 'url': self.url } if self.country is not None: params['country'] = self.country for name, value in self.headers.items(): params['headers[%s]' % name] = value if self.webhook is not None: params['webhook_name'] = self.webhook if self.timeout is not None: params['timeout'] = self.timeout if self.extract is not None: params['extract'] = base64.urlsafe_b64encode(json.dumps(self.extract).encode('utf-8')).decode('utf-8') if self.cost_budget is not None: params['cost_budget'] = self.cost_budget if self.proxified_response is not None: params['proxified_response'] = self._bool_to_http(self.proxified_response) if self.render_js is True: params['render_js'] = self._bool_to_http(self.render_js) if self.wait_for_selector is not None: params['wait_for_selector'] = self.wait_for_selector if self.js: params['js'] = base64.urlsafe_b64encode(self.js.encode('utf-8')).decode('utf-8') if self.js_scenario: params['js_scenario'] = base64.urlsafe_b64encode(json.dumps(self.js_scenario).encode('utf-8')).decode('utf-8') if self.rendering_wait: params['rendering_wait'] = self.rendering_wait if self.rendering_stage: params['rendering_stage'] = self.rendering_stage if self.screenshots is not None: for name, element in self.screenshots.items(): params['screenshots[%s]' % name] = element if self.screenshot_flags is not None: self.screenshot_flags = [ScreenshotFlag(flag) for flag in self.screenshot_flags] params["screenshot_flags"] = ",".join(flag.value for flag in self.screenshot_flags) else: if self.screenshot_flags is not None: logging.warning('Params "screenshot_flags" is ignored. Works only if screenshots is enabled') if self.auto_scroll is True: params['auto_scroll'] = self._bool_to_http(self.auto_scroll) else: if self.wait_for_selector is not None: logging.warning('Params "wait_for_selector" is ignored. Works only if render_js is enabled') if self.screenshots: logging.warning('Params "screenshots" is ignored. Works only if render_js is enabled') if self.js_scenario: logging.warning('Params "js_scenario" is ignored. Works only if render_js is enabled') if self.js: logging.warning('Params "js" is ignored. Works only if render_js is enabled') if self.rendering_wait: logging.warning('Params "rendering_wait" is ignored. Works only if render_js is enabled') # The emitted key stays `asp`. Published SDK versions are immutable and # upgraded per installation, so emitting `unblocker` against an API # deployment that has not learned it yet would silently drop a paid # feature: the scrape succeeds, is billed, and returns a blocked page. if self.asp is True: params['asp'] = self._bool_to_http(self.asp) if self.retry is False: params['retry'] = self._bool_to_http(self.retry) if self.cache is True: params['cache'] = self._bool_to_http(self.cache) if self.cache_clear is True: params['cache_clear'] = self._bool_to_http(self.cache_clear) if self.cache_ttl is not None: params['cache_ttl'] = self.cache_ttl else: if self.cache_clear is True: logging.warning('Params "cache_clear" is ignored. Works only if cache is enabled') if self.cache_ttl is not None: logging.warning('Params "cache_ttl" is ignored. Works only if cache is enabled') if self.dns is True: params['dns'] = self._bool_to_http(self.dns) if self.ssl is True: params['ssl'] = self._bool_to_http(self.ssl) if self.tags: params['tags'] = ','.join(self.tags) if self.format: params['format'] = Format(self.format).value if self.format_options: params['format'] += ':' + ','.join(FormatOption(option).value for option in self.format_options) if self.extraction_template and self.extraction_ephemeral_template: raise ScrapeConfigError('You cannot pass both parameters extraction_template and extraction_ephemeral_template. You must choose') if self.extraction_template: params['extraction_template'] = 'persistent:' + self.extraction_template if self.extraction_ephemeral_template: self.extraction_ephemeral_template = json.dumps(self.extraction_ephemeral_template) params['extraction_template'] = 'ephemeral:' + urlsafe_b64encode(self.extraction_ephemeral_template.encode('utf-8')).decode('utf-8') if self.extraction_prompt: params['extraction_prompt'] = quote_plus(self.extraction_prompt) if self.extraction_model: params['extraction_model'] = self.extraction_model if self.correlation_id: params['correlation_id'] = self.correlation_id if self.session: params['session'] = self.session if self.session_sticky_proxy is not None: params['session_sticky_proxy'] = self._bool_to_http(self.session_sticky_proxy) else: if self.session_sticky_proxy: logging.warning('Params "session_sticky_proxy" is ignored. Works only if session is enabled') if self.debug is True: params['debug'] = self._bool_to_http(self.debug) if self.proxy_pool is not None: params['proxy_pool'] = self.proxy_pool if self.lang is not None: params['lang'] = ','.join(self.lang) if self.os is not None: params['os'] = self.os if self.browser_brand is not None: params['browser_brand'] = self.browser_brand if self.geolocation is not None: params['geolocation'] = self.geolocation return params def to_dict(self) ‑> Dict-
Expand source code
def to_dict(self) -> Dict: """ Export the ScrapeConfig instance to a plain dictionary. Useful for JSON-serialization or other external storage. """ return { 'url': self.url, 'retry': self.retry, 'method': self.method, 'country': self.country, 'render_js': self.render_js, 'cache': self.cache, 'cache_clear': self.cache_clear, 'ssl': self.ssl, 'dns': self.dns, 'asp': self.asp, 'debug': self.debug, 'raise_on_upstream_error': self.raise_on_upstream_error, 'cache_ttl': self.cache_ttl, 'proxy_pool': self.proxy_pool, 'session': self.session, 'tags': list(self.tags), 'format': Format(self.format).value if self.format else None, 'format_options': [FormatOption(option).value for option in self.format_options] if self.format_options else None, 'extraction_template': self.extraction_template, 'extraction_ephemeral_template': self.extraction_ephemeral_template, 'extraction_prompt': self.extraction_prompt, 'extraction_model': self.extraction_model, 'correlation_id': self.correlation_id, 'cookies': CaseInsensitiveDict(self.cookies), 'body': self.body, 'data': None if self.body else self.data, 'headers': CaseInsensitiveDict(self.headers), 'js': self.js, 'rendering_wait': self.rendering_wait, 'wait_for_selector': self.wait_for_selector, 'session_sticky_proxy': self.session_sticky_proxy, 'screenshots': self.screenshots, 'screenshot_flags': [ScreenshotFlag(flag).value for flag in self.screenshot_flags] if self.screenshot_flags else None, 'webhook': self.webhook, 'timeout': self.timeout, 'js_scenario': self.js_scenario, 'extract': self.extract, 'lang': self.lang, 'os': self.os, 'auto_scroll': self.auto_scroll, 'cost_budget': self.cost_budget, 'browser_brand': self.browser_brand, }Export the ScrapeConfig instance to a plain dictionary. Useful for JSON-serialization or other external storage.
class ScraperAPI-
Expand source code
class ScraperAPI: MONITORING_DATA_FORMAT_STRUCTURED = 'structured' MONITORING_DATA_FORMAT_PROMETHEUS = 'prometheus' MONITORING_PERIOD_SUBSCRIPTION = 'subscription' MONITORING_PERIOD_LAST_7D = 'last7d' MONITORING_PERIOD_LAST_24H = 'last24h' MONITORING_PERIOD_LAST_1H = 'last1h' MONITORING_PERIOD_LAST_5m = 'last5m' MONITORING_ACCOUNT_AGGREGATION = 'account' MONITORING_PROJECT_AGGREGATION = 'project' MONITORING_TARGET_AGGREGATION = 'target'Class variables
var MONITORING_ACCOUNT_AGGREGATIONvar MONITORING_DATA_FORMAT_PROMETHEUSvar MONITORING_DATA_FORMAT_STRUCTUREDvar MONITORING_PERIOD_LAST_1Hvar MONITORING_PERIOD_LAST_24Hvar MONITORING_PERIOD_LAST_5mvar MONITORING_PERIOD_LAST_7Dvar MONITORING_PERIOD_SUBSCRIPTIONvar MONITORING_PROJECT_AGGREGATIONvar MONITORING_TARGET_AGGREGATION
class ScrapflyAspError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class ScrapflyAspError(ScraperAPIError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.ScraperAPIError
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class ScrapflyUnblockerError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class ScrapflyAspError(ScraperAPIError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.ScraperAPIError
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class ScrapflyClient (key: str,
host: str = 'https://api.scrapfly.io',
verify=True,
debug: bool = False,
max_concurrency: int = 1,
connect_timeout: int = 30,
web_scraping_api_read_timeout: int = 160,
extraction_api_read_timeout: int = 35,
screenshot_api_read_timeout: int = 60,
read_timeout: int = 30,
default_read_timeout: int = 30,
reporter: Callable | None = None,
cloud_browser_host: str | None = None,
**kwargs)-
Expand source code
class ScrapflyClient(ScheduleClientMixin): HOST = 'https://api.scrapfly.io' CLOUD_BROWSER_HOST = 'wss://browser.scrapfly.io' CLOUD_BROWSER_API_HOST = 'https://browser.scrapfly.io' DEFAULT_CONNECT_TIMEOUT = 30 DEFAULT_READ_TIMEOUT = 30 DEFAULT_WEBSCRAPING_API_READ_TIMEOUT = 160 # 155 real DEFAULT_SCREENSHOT_API_READ_TIMEOUT = 60 # 30 real DEFAULT_EXTRACTION_API_READ_TIMEOUT = 35 # 30 real DEFAULT_CRAWLER_API_READ_TIMEOUT = 30 # A search fans out over every requested crawl before answering. DEFAULT_CRAWLER_SEARCH_API_READ_TIMEOUT = 60 # Retrieval plus generation. The whole exchange is budgeted under the # API's own 165s upstream ceiling, so a longer client read is pointless. DEFAULT_CRAWLER_PROMPT_API_READ_TIMEOUT = 180 host:str key:str max_concurrency:int verify:bool debug:bool distributed_mode:bool connect_timeout:int web_scraping_api_read_timeout:int screenshot_api_read_timeout:int extraction_api_read_timeout:int monitoring_api_read_timeout:int default_read_timeout:int brotli: bool reporter:Reporter version:str # @deprecated read_timeout:int CONCURRENCY_AUTO = 'auto' # retrieve the allowed concurrency from your account DATETIME_FORMAT = '%Y-%m-%d %H:%M:%S' def __init__( self, key: str, host: str = HOST, verify=True, debug: bool = False, max_concurrency:int=1, connect_timeout:int = DEFAULT_CONNECT_TIMEOUT, web_scraping_api_read_timeout: int = DEFAULT_WEBSCRAPING_API_READ_TIMEOUT, extraction_api_read_timeout: int = DEFAULT_EXTRACTION_API_READ_TIMEOUT, screenshot_api_read_timeout: int = DEFAULT_SCREENSHOT_API_READ_TIMEOUT, # @deprecated read_timeout:int = DEFAULT_READ_TIMEOUT, default_read_timeout:int = DEFAULT_READ_TIMEOUT, reporter:Optional[Callable]=None, cloud_browser_host: Optional[str] = None, **kwargs ): if host[-1] == '/': # remove last '/' if exists host = host[:-1] if 'distributed_mode' in kwargs: warnings.warn("distributed mode is deprecated and will be remove the next version -" " user should handle themself the session name based on the concurrency", DeprecationWarning, stacklevel=2 ) if 'brotli' in kwargs: warnings.warn("brotli arg is deprecated and will be remove the next version - " "brotli is disabled by default", DeprecationWarning, stacklevel=2 ) self.version = __version__ self.host = host self.key = key self.verify = verify self.cloud_browser_host = cloud_browser_host or self.CLOUD_BROWSER_HOST self.cloud_browser_api_host = cloud_browser_host.replace('wss://', 'https://') if cloud_browser_host else self.CLOUD_BROWSER_API_HOST self.debug = debug self.connect_timeout = connect_timeout self.web_scraping_api_read_timeout = web_scraping_api_read_timeout self.screenshot_api_read_timeout = screenshot_api_read_timeout self.extraction_api_read_timeout = extraction_api_read_timeout self.monitoring_api_read_timeout = default_read_timeout self.default_read_timeout = default_read_timeout # @deprecated self.read_timeout = default_read_timeout self.max_concurrency = max_concurrency self.body_handler = ResponseBodyHandler(use_brotli=False) self.async_executor = ThreadPoolExecutor() self.http_session = None if not self.verify and not self.HOST.endswith('.local'): urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning) if self.debug is True: http.client.HTTPConnection.debuglevel = 5 if reporter is None: from .reporter import NoopReporter reporter = NoopReporter() self.reporter = Reporter(reporter) @property def ua(self) -> str: return 'ScrapflySDK/%s (Python %s, %s, %s)' % ( self.version, platform.python_version(), platform.uname().system, platform.uname().machine ) @cached_property def _http_handler(self): return partial(self.http_session.request if self.http_session else requests.request) @property def http(self): return self._http_handler def _scrape_request(self, scrape_config:ScrapeConfig): return { 'method': scrape_config.method, 'url': self.host + '/scrape', 'data': scrape_config.body, 'verify': self.verify, 'timeout': (self.connect_timeout, self.web_scraping_api_read_timeout), 'headers': { # When method has a body (POST/PUT/PATCH) AND the caller # explicitly set a Content-Type, forward it. Otherwise fall # back to the body_handler default so we don't KeyError on # callers who omit the header (e.g. simple PUT "test-body"). 'content-type': ( scrape_config.headers.get('content-type', self.body_handler.content_type) if scrape_config.method in ['POST', 'PUT', 'PATCH'] else self.body_handler.content_type ), 'accept-encoding': self.body_handler.content_encoding, 'accept': self.body_handler.accept, 'user-agent': self.ua }, 'params': scrape_config.to_api_params(key=self.key) } def _screenshot_request(self, screenshot_config:ScreenshotConfig): return { 'method': 'GET', 'url': self.host + '/screenshot', 'timeout': (self.connect_timeout, self.screenshot_api_read_timeout), 'verify': self.verify, 'headers': { 'accept-encoding': self.body_handler.content_encoding, 'accept': self.body_handler.accept, 'user-agent': self.ua }, 'params': screenshot_config.to_api_params(key=self.key) } def _extraction_request(self, extraction_config:ExtractionConfig): headers = { 'content-type': extraction_config.content_type, 'accept-encoding': self.body_handler.content_encoding, 'content-encoding': extraction_config.document_compression_format if extraction_config.document_compression_format else None, 'accept': self.body_handler.accept, 'user-agent': self.ua } if extraction_config.document_compression_format: headers['content-encoding'] = extraction_config.document_compression_format.value return { 'method': 'POST', 'url': self.host + '/extraction', 'data': extraction_config.body, 'timeout': (self.connect_timeout, self.extraction_api_read_timeout), 'verify': self.verify, 'headers': headers, 'params': extraction_config.to_api_params(key=self.key) } def account(self) -> Union[str, Dict]: response = self._http_handler( method='GET', url=self.host + '/account', params={'key': self.key}, verify=self.verify, headers={ 'accept-encoding': self.body_handler.content_encoding, 'accept': self.body_handler.accept, 'user-agent': self.ua }, ) response.raise_for_status() if self.body_handler.support(response.headers): return self.body_handler(response.content, response.headers['content-type']) return response.content.decode('utf-8') def classify( self, url: str, status_code: int, headers: Optional[Dict[str, str]] = None, body: Optional[str] = None, method: str = "GET", ) -> ClassifyResult: """Classify an already-fetched HTTP response for anti-bot blocking. Runs the same 80+ shield pipeline used by live Scrapfly scrapes against a response you already have (from your own proxy, cache, etc). 1 API credit per call. See https://scrapfly.io/docs/scrape-api/classify for the full contract. """ if not url: raise ContentError("classify: url is required") if not (100 <= int(status_code) <= 599): raise ContentError( "classify: status_code must be a valid HTTP status in [100, 599]" ) payload: Dict[str, Any] = { "url": url, "status_code": int(status_code), "method": method or "GET", } if headers: payload["headers"] = {str(k): str(v) for k, v in headers.items()} if body is not None: payload["body"] = body response = self._http_handler( method="POST", url=self.host + "/classify", params={"key": self.key}, json=payload, verify=self.verify, headers={ "accept-encoding": self.body_handler.content_encoding, "accept": self.body_handler.accept, "user-agent": self.ua, "content-type": "application/json", }, ) response.raise_for_status() if self.body_handler.support(response.headers): data = self.body_handler(response.content, response.headers["content-type"]) else: import json as _json data = _json.loads(response.content.decode("utf-8")) return ClassifyResult.from_dict(data) # ── Monitoring API (Enterprise+ plan only) ────────────────────── # The Monitoring API exposes per-product aggregates and per-target # timeseries. Web Scraping / Screenshot / Extraction / Crawler share # one shape (request-based) but live under different URL prefixes; # Cloud Browser is session-based and exposes a distinct shape. # See https://scrapfly.io/docs/monitoring#api @staticmethod def _format_monitoring_dt(dt:datetime.datetime) -> str: """Format a datetime in UTC as YYYY-MM-DD HH:MM:SS for the Monitoring API. Naive datetimes are assumed to be UTC; aware datetimes are converted via astimezone(timezone.utc) so SDK behavior matches the Go/TypeScript SDKs (which always emit UTC).""" if dt.tzinfo is not None: dt = dt.astimezone(datetime.timezone.utc) return dt.strftime('%Y-%m-%d %H:%M:%S') def _monitoring_request(self, path:str, params:dict): """Internal helper. Issues a GET against the Monitoring API and decodes the response via the standard body_handler.""" response = self._http_handler( method='GET', url=self.host + path, params=params, timeout=(self.connect_timeout, self.monitoring_api_read_timeout), verify=self.verify, headers={ 'accept-encoding': self.body_handler.content_encoding, 'accept': self.body_handler.accept, 'user-agent': self.ua }, ) response.raise_for_status() if self.body_handler.support(response.headers): return self.body_handler(response.content, response.headers['content-type']) return response.content.decode('utf-8') def _build_metrics_params( self, format:str, period:Optional[str], aggregation:Optional[List[MonitoringAggregation]], include_webhook:bool, ) -> dict: params = {'key': self.key, 'format': format} if period is not None: params['period'] = period if aggregation is not None: params['aggregation'] = ','.join(aggregation) if include_webhook: params['include_webhook'] = 'true' return params def _build_target_params( self, domain:str, group_subdomain:bool, period:Optional[MonitoringTargetPeriod], start:Optional[datetime.datetime], end:Optional[datetime.datetime], include_webhook:bool, ) -> dict: if (start is not None and end is None) or (start is None and end is not None): raise ValueError('You must provide both start and end date') params = { 'key': self.key, 'domain': domain, 'group_subdomain': group_subdomain, } if start is not None and end is not None: params['start'] = self._format_monitoring_dt(start) params['end'] = self._format_monitoring_dt(end) period = None params['period'] = period if include_webhook: params['include_webhook'] = 'true' return params # ── Web Scraping API ───────────────────────────────────────────── def get_monitoring_metrics( self, format:str=ScraperAPI.MONITORING_DATA_FORMAT_STRUCTURED, period:Optional[str]=None, aggregation:Optional[List[MonitoringAggregation]]=None, include_webhook:bool=False, ): return self._monitoring_request( '/scrape/monitoring/metrics', self._build_metrics_params(format, period, aggregation, include_webhook), ) def get_monitoring_target_metrics( self, domain:str, group_subdomain:bool=False, period:Optional[MonitoringTargetPeriod]=ScraperAPI.MONITORING_PERIOD_LAST_24H, start:Optional[datetime.datetime]=None, end:Optional[datetime.datetime]=None, include_webhook:bool=False, ): return self._monitoring_request( '/scrape/monitoring/metrics/target', self._build_target_params(domain, group_subdomain, period, start, end, include_webhook), ) # ── Screenshot API ─────────────────────────────────────────────── def get_screenshot_monitoring_metrics( self, format:str=ScraperAPI.MONITORING_DATA_FORMAT_STRUCTURED, period:Optional[str]=None, aggregation:Optional[List[MonitoringAggregation]]=None, include_webhook:bool=False, ): return self._monitoring_request( '/screenshot/monitoring/metrics', self._build_metrics_params(format, period, aggregation, include_webhook), ) def get_screenshot_monitoring_target_metrics( self, domain:str, group_subdomain:bool=False, period:Optional[MonitoringTargetPeriod]=ScraperAPI.MONITORING_PERIOD_LAST_24H, start:Optional[datetime.datetime]=None, end:Optional[datetime.datetime]=None, include_webhook:bool=False, ): return self._monitoring_request( '/screenshot/monitoring/metrics/target', self._build_target_params(domain, group_subdomain, period, start, end, include_webhook), ) # ── Extraction API ─────────────────────────────────────────────── def get_extraction_monitoring_metrics( self, format:str=ScraperAPI.MONITORING_DATA_FORMAT_STRUCTURED, period:Optional[str]=None, aggregation:Optional[List[MonitoringAggregation]]=None, include_webhook:bool=False, ): return self._monitoring_request( '/extraction/monitoring/metrics', self._build_metrics_params(format, period, aggregation, include_webhook), ) def get_extraction_monitoring_target_metrics( self, domain:str, group_subdomain:bool=False, period:Optional[MonitoringTargetPeriod]=ScraperAPI.MONITORING_PERIOD_LAST_24H, start:Optional[datetime.datetime]=None, end:Optional[datetime.datetime]=None, include_webhook:bool=False, ): return self._monitoring_request( '/extraction/monitoring/metrics/target', self._build_target_params(domain, group_subdomain, period, start, end, include_webhook), ) # ── Crawler API ────────────────────────────────────────────────── def get_crawler_monitoring_metrics( self, format:str=ScraperAPI.MONITORING_DATA_FORMAT_STRUCTURED, period:Optional[str]=None, aggregation:Optional[List[MonitoringAggregation]]=None, include_webhook:bool=False, ): return self._monitoring_request( '/crawl/monitoring/metrics', self._build_metrics_params(format, period, aggregation, include_webhook), ) def get_crawler_monitoring_target_metrics( self, domain:str, group_subdomain:bool=False, period:Optional[MonitoringTargetPeriod]=ScraperAPI.MONITORING_PERIOD_LAST_24H, start:Optional[datetime.datetime]=None, end:Optional[datetime.datetime]=None, include_webhook:bool=False, ): return self._monitoring_request( '/crawl/monitoring/metrics/target', self._build_target_params(domain, group_subdomain, period, start, end, include_webhook), ) # ── Cloud Browser API (session-based, distinct shape) ──────────── def get_browser_monitoring_metrics( self, period:Optional[str]=None, proxy_pool:Optional[str]=None, start:Optional[datetime.datetime]=None, end:Optional[datetime.datetime]=None, ): if (start is not None and end is None) or (start is None and end is not None): raise ValueError('You must provide both start and end date') params:dict = {'key': self.key} if start is not None and end is not None: params['start'] = self._format_monitoring_dt(start) params['end'] = self._format_monitoring_dt(end) elif period is not None: params['period'] = period if proxy_pool is not None: params['proxy_pool'] = proxy_pool return self._monitoring_request('/browser/monitoring/metrics', params) def get_browser_monitoring_timeseries( self, period:Optional[str]=None, proxy_pool:Optional[str]=None, start:Optional[datetime.datetime]=None, end:Optional[datetime.datetime]=None, ): if (start is not None and end is None) or (start is None and end is not None): raise ValueError('You must provide both start and end date') params:dict = {'key': self.key} if start is not None and end is not None: params['start'] = self._format_monitoring_dt(start) params['end'] = self._format_monitoring_dt(end) elif period is not None: params['period'] = period if proxy_pool is not None: params['proxy_pool'] = proxy_pool return self._monitoring_request('/browser/monitoring/metrics/timeseries', params) def resilient_scrape( self, scrape_config:ScrapeConfig, retry_on_errors:Optional[Set[Exception]]=None, retry_on_status_code:Optional[List[int]]=None, tries: int = 5, delay: int = 20, ) -> ScrapeApiResponse: if retry_on_errors is None: retry_on_errors = {ScrapflyError} assert isinstance(retry_on_errors, set), 'retry_on_errors is not a set()' @backoff.on_exception(backoff.expo, exception=tuple(retry_on_errors), max_tries=tries, max_time=delay) def inner() -> ScrapeApiResponse: try: return self.scrape(scrape_config=scrape_config) except (UpstreamHttpClientError, UpstreamHttpServerError) as e: if retry_on_status_code is not None and e.api_response: if e.api_response.upstream_status_code in retry_on_status_code: raise e else: return e.api_response raise e return inner() def open(self): if self.http_session is None: self.http_session = Session() self.http_session.verify = self.verify self.http_session.timeout = (self.connect_timeout, self.default_read_timeout) self.http_session.params['key'] = self.key self.http_session.headers['accept-encoding'] = self.body_handler.content_encoding self.http_session.headers['accept'] = self.body_handler.accept self.http_session.headers['user-agent'] = self.ua def close(self): if self.http_session is not None: self.http_session.close() self.http_session = None # The executor is created in __init__ and owns worker threads that # outlive the HTTP session; shutting it down here prevents thread # leaks for callers that reuse the client across open()/close() # cycles or rely on GC to reclaim it. if self.async_executor is not None: self.async_executor.shutdown(wait=False) self.async_executor = None def __enter__(self) -> 'ScrapflyClient': self.open() return self def __exit__(self, exc_type, exc_val, exc_tb): self.close() async def async_scrape(self, scrape_config:ScrapeConfig, loop:Optional[AbstractEventLoop]=None) -> ScrapeApiResponse: if loop is None: loop = asyncio.get_running_loop() return await loop.run_in_executor(self.async_executor, self.scrape, scrape_config) async def concurrent_scrape(self, scrape_configs:List[ScrapeConfig], concurrency:Optional[int]=None): if concurrency is None: concurrency = self.max_concurrency elif concurrency == self.CONCURRENCY_AUTO: concurrency = self.account()['subscription']['max_concurrency'] loop = asyncio.get_running_loop() processing_tasks = [] results = [] processed_tasks = 0 expected_tasks = len(scrape_configs) def scrape_done_callback(task:Task): nonlocal processed_tasks try: if task.cancelled() is True: return error = task.exception() if error is not None: results.append(error) else: results.append(task.result()) finally: processing_tasks.remove(task) processed_tasks += 1 while scrape_configs or results or processing_tasks: logger.info("Scrape %d/%d - %d running" % (processed_tasks, expected_tasks, len(processing_tasks))) if scrape_configs: if len(processing_tasks) < concurrency: # @todo handle backpressure for _ in range(0, concurrency - len(processing_tasks)): try: scrape_config = scrape_configs.pop() except IndexError: break scrape_config.raise_on_upstream_error = False task = loop.create_task(self.async_scrape(scrape_config=scrape_config, loop=loop)) processing_tasks.append(task) task.add_done_callback(scrape_done_callback) for _ in results: result = results.pop() yield result await asyncio.sleep(.5) logger.debug("Scrape %d/%d - %d running" % (processed_tasks, expected_tasks, len(processing_tasks))) @backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def scrape(self, scrape_config:ScrapeConfig, no_raise:bool=False) -> ScrapeApiResponse: """ Scrape a website :param scrape_config: ScrapeConfig :param no_raise: bool - if True, do not raise exception on error while the api response is a ScrapflyError for seamless integration :return: ScrapeApiResponse If you use no_raise=True, make sure to check the api_response.scrape_result.error attribute to handle the error. If the error is not none, you will get the following structure for example 'error': { 'code': 'ERR::ASP::SHIELD_PROTECTION_FAILED', 'message': 'The ASP shield failed to solve the challenge against the anti scrapping protection - heuristic_engine bypass failed, please retry in few seconds', 'retryable': False, 'http_code': 422, 'links': { 'Checkout ASP documentation': 'https://scrapfly.io/docs/scrape-api/anti-scraping-protection#maximize_success_rate', 'Related Error Doc': 'https://scrapfly.io/docs/scrape-api/error/ERR::ASP::SHIELD_PROTECTION_FAILED' } } """ try: logger.debug('--> %s Scrapping %s' % (scrape_config.method, scrape_config.url)) request_data = self._scrape_request(scrape_config=scrape_config) response = self._http_handler(**request_data) if scrape_config.proxified_response is True: # Proxified mode: the API returns the raw upstream response # (target's status, headers, body) instead of the JSON # envelope. Error restoration: if X-Scrapfly-Reject-Code is # present, the scrape failed and the SDK must raise a typed # error with the code/message/retryable from the headers. reject_code = response.headers.get('X-Scrapfly-Reject-Code') if reject_code: from scrapfly.errors import HttpError reject_desc = response.headers.get('X-Scrapfly-Reject-Description', '') reject_retryable = response.headers.get('X-Scrapfly-Reject-Retryable', 'false').lower() == 'true' retry_after = None if reject_retryable: try: retry_after = int(response.headers.get('Retry-After', '0')) except (ValueError, TypeError): retry_after = None raise HttpError( request=response.request, response=response, code=reject_code, http_status_code=response.status_code, message=reject_desc, is_retryable=reject_retryable, retry_delay=retry_after, ) self.reporter.report(scrape_api_response=None) return response scrape_api_response = self._handle_response(response=response, scrape_config=scrape_config) self.reporter.report(scrape_api_response=scrape_api_response) return scrape_api_response except BaseException as e: self.reporter.report(error=e) if no_raise and isinstance(e, ScrapflyError) and e.api_response is not None: return e.api_response raise e async def async_screenshot(self, screenshot_config:ScreenshotConfig, loop:Optional[AbstractEventLoop]=None) -> ScreenshotApiResponse: if loop is None: loop = asyncio.get_running_loop() return await loop.run_in_executor(self.async_executor, self.screenshot, screenshot_config) @backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def screenshot(self, screenshot_config:ScreenshotConfig, no_raise:bool=False) -> ScreenshotApiResponse: """ Take a screenshot :param screenshot_config: ScrapeConfig :param no_raise: bool - if True, do not raise exception on error while the screenshot api response is a ScrapflyError for seamless integration :return: str If you use no_raise=True, make sure to check the screenshot_api_response.error attribute to handle the error. If the error is not none, you will get the following structure for example 'error': { 'code': 'ERR::SCREENSHOT::UNABLE_TO_TAKE_SCREENSHOT', 'message': 'For some reason we were unable to take the screenshot', 'http_code': 422, 'links': { 'Checkout the related doc: https://scrapfly.io/docs/screenshot-api/error/ERR::SCREENSHOT::UNABLE_TO_TAKE_SCREENSHOT' } } """ try: logger.debug('--> %s Screenshoting' % (screenshot_config.url)) request_data = self._screenshot_request(screenshot_config=screenshot_config) response = self._http_handler(**request_data) screenshot_api_response = self._handle_screenshot_response(response=response, screenshot_config=screenshot_config) return screenshot_api_response except BaseException as e: self.reporter.report(error=e) if no_raise and isinstance(e, ScrapflyError) and e.api_response is not None: return e.api_response raise e async def async_extraction(self, extraction_config:ExtractionConfig, loop:Optional[AbstractEventLoop]=None) -> ExtractionApiResponse: if loop is None: loop = asyncio.get_running_loop() return await loop.run_in_executor(self.async_executor, self.extract, extraction_config) @backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def extract(self, extraction_config:ExtractionConfig, no_raise:bool=False) -> ExtractionApiResponse: """ Extract structured data from text content :param extraction_config: ExtractionConfig :param no_raise: bool - if True, do not raise exception on error while the extraction api response is a ScrapflyError for seamless integration :return: str If you use no_raise=True, make sure to check the extraction_api_response.error attribute to handle the error. If the error is not none, you will get the following structure for example 'error': { 'code': 'ERR::EXTRACTION::CONTENT_TYPE_NOT_SUPPORTED', 'message': 'The content type of the response is not supported for extraction', 'http_code': 422, 'links': { 'Checkout the related doc: https://scrapfly.io/docs/extraction-api/error/ERR::EXTRACTION::CONTENT_TYPE_NOT_SUPPORTED' } } """ try: logger.debug('--> %s Extracting data from' % (extraction_config.content_type)) request_data = self._extraction_request(extraction_config=extraction_config) response = self._http_handler(**request_data) extraction_api_response = self._handle_extraction_response(response=response, extraction_config=extraction_config) return extraction_api_response except BaseException as e: self.reporter.report(error=e) if no_raise and isinstance(e, ScrapflyError) and e.api_response is not None: return e.api_response raise e def _handle_response(self, response:Response, scrape_config:ScrapeConfig) -> ScrapeApiResponse: try: api_response = self._handle_api_response( response=response, scrape_config=scrape_config, raise_on_upstream_error=scrape_config.raise_on_upstream_error ) if scrape_config.method == 'HEAD': logger.debug('<-- [%s %s] %s | %ss' % ( api_response.response.status_code, api_response.response.reason, api_response.response.request.url, 0 )) else: logger.debug('<-- [%s %s] %s | %ss' % ( api_response.result['result']['status_code'], api_response.result['result']['reason'], api_response.result['config']['url'], api_response.result['result']['duration']) ) logger.debug('Log url: %s' % api_response.result['result']['log_url']) return api_response except UpstreamHttpError as e: logger.critical(e.api_response.error_message) raise except HttpError as e: if e.api_response is not None: logger.critical(e.api_response.error_message) else: logger.critical(e.message) raise except ScrapflyError as e: logger.critical('<-- %s | Docs: %s' % (str(e), e.documentation_url)) raise def _handle_screenshot_response(self, response:Response, screenshot_config:ScreenshotConfig) -> ScreenshotApiResponse: try: api_response = self._handle_screenshot_api_response( response=response, screenshot_config=screenshot_config, raise_on_upstream_error=screenshot_config.raise_on_upstream_error ) return api_response except UpstreamHttpError as e: logger.critical(e.api_response.error_message) raise except HttpError as e: if e.api_response is not None: logger.critical(e.api_response.error_message) else: logger.critical(e.message) raise except ScrapflyError as e: logger.critical('<-- %s | Docs: %s' % (str(e), e.documentation_url)) raise def _handle_extraction_response(self, response:Response, extraction_config:ExtractionConfig) -> ExtractionApiResponse: try: api_response = self._handle_extraction_api_response( response=response, extraction_config=extraction_config, raise_on_upstream_error=extraction_config.raise_on_upstream_error ) return api_response except UpstreamHttpError as e: logger.critical(e.api_response.error_message) raise except HttpError as e: if e.api_response is not None: logger.critical(e.api_response.error_message) else: logger.critical(e.message) raise except ScrapflyError as e: logger.critical('<-- %s | Docs: %s' % (str(e), e.documentation_url)) raise @backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def scrape_batch( self, scrape_configs: List[ScrapeConfig], format: Optional[Literal['json', 'msgpack']] = None, ) -> Iterator[Tuple[str, Union[ScrapeApiResponse, ScrapflyError]]]: """ Scrape up to 100 URLs in one batch request and stream results back as each scrape completes. Iterator yields ``(correlation_id, result)`` tuples where ``result`` is either a :class:`ScrapeApiResponse` on success or a :class:`ScrapflyError` on per-scrape failure. Results arrive **out of order** — whichever scrape finishes first is yielded first. Use ``correlation_id`` (set on every ``ScrapeConfig``) to match parts back to the originating config on the client side. Every config MUST carry a unique ``correlation_id``; a missing/duplicate value is detected client-side before the batch is sent. :param format: wire format for per-part response bodies. Defaults to the SDK's negotiated format (``msgpack`` when the ``msgpack`` package is installed, ``json`` otherwise). Pass ``'json'`` or ``'msgpack'`` to override. """ from .batch import ( iter_batch_parts, decode_part_body, is_api_error_part, error_from_api_error_part, _build_proxified_response_from_part, ) if not scrape_configs: raise ScrapflyError( "scrape_batch: configs list is empty", code="ERR::SCRAPE::BATCH_CONFIG", http_status_code=400, ) if len(scrape_configs) > 100: raise ScrapflyError( f"scrape_batch: max 100 configs per batch (got {len(scrape_configs)})", code="ERR::SCRAPE::BATCH_CONFIG", http_status_code=400, ) seen_correlations: Dict[str, int] = {} body_configs: List[Dict[str, Any]] = [] config_by_correlation: Dict[str, ScrapeConfig] = {} for idx, cfg in enumerate(scrape_configs): if not getattr(cfg, "correlation_id", None): raise ScrapflyError( f"scrape_batch: configs[{idx}] is missing correlation_id " "(required for matching streamed parts)", code="ERR::SCRAPE::BATCH_CONFIG", http_status_code=422, ) if cfg.correlation_id in seen_correlations: raise ScrapflyError( f"scrape_batch: correlation_id {cfg.correlation_id!r} reused by " f"configs[{seen_correlations[cfg.correlation_id]}] and configs[{idx}]", code="ERR::SCRAPE::BATCH_CONFIG", http_status_code=422, ) seen_correlations[cfg.correlation_id] = idx config_by_correlation[cfg.correlation_id] = cfg # Drop `key` (batch key goes in the URL); pass everything # else as a flat query-param dict. The server feeds each # entry through NewScrapeConfigFromRequest identically to # a /scrape call, so the wire contract is identical. params = cfg.to_api_params(key=self.key) params.pop("key", None) body_configs.append(params) import json as _json payload = _json.dumps({"configs": body_configs}).encode("utf-8") if format == 'msgpack': accept_header = 'application/msgpack' elif format == 'json': accept_header = 'application/json' else: accept_header = self.body_handler.accept request = { "method": "POST", "url": self.host + "/scrape/batch", "params": {"key": self.key}, "data": payload, "headers": { "content-type": "application/json", "accept-encoding": self.body_handler.content_encoding, "accept": accept_header, "user-agent": self.ua, }, "timeout": (self.connect_timeout, self.web_scraping_api_read_timeout), "verify": self.verify, "stream": True, } # Own the session for the life of the streaming batch so its # connection pool closes whether the generator is fully consumed, # errors mid-stream, or is abandoned (finally runs on GC/close()). batch_session = requests.Session() batch_session.verify = self.verify try: response = batch_session.request( method=request["method"], url=request["url"], params=request["params"], data=request["data"], headers=request["headers"], timeout=request["timeout"], stream=request["stream"], ) if response.status_code != 200: # Batch-level error (plan gate, validation, insufficient # concurrency, etc.). Response is a single JSON body, not # multipart. try: body = response.json() except Exception: body = {"message": response.text, "code": "ERR::API::INTERNAL_ERROR"} err_code = body.get("code", "ERR::API::INTERNAL_ERROR") err_msg = body.get("message", "") or body.get("reason", "") retry_after = None try: retry_after = int(response.headers.get("Retry-After", "0")) or None except (TypeError, ValueError): pass raise HttpError( request=response.request, response=response, code=err_code, http_status_code=response.status_code, message=err_msg, is_retryable=body.get("retryable", False), retry_delay=retry_after, ) for part_headers, part_body in iter_batch_parts(response): correlation_id = part_headers.get("x-scrapfly-correlation-id", "") cfg = config_by_correlation.get(correlation_id, scrape_configs[0]) # Proxified-response parts: the part body is the raw # upstream bytes, not a JSON envelope. Surface a native # requests.Response synthesized from the part headers + # body so callers get the same shape as a single # proxified scrape. if part_headers.get("x-scrapfly-proxified") == "true": try: prox_response = _build_proxified_response_from_part( part_headers, part_body, originating_request=response.request, ) except Exception as prox_err: yield correlation_id, ScrapflyError( f"scrape_batch: failed to build proxified response for correlation_id={correlation_id!r}: {prox_err}", code="ERR::API::INTERNAL_ERROR", http_status_code=500, ) continue yield correlation_id, prox_response continue # EncoderError subclasses BaseException — catch it explicitly. try: parsed = decode_part_body(part_headers, part_body, self.body_handler) except (EncoderError, Exception) as decode_err: yield correlation_id, ScrapflyError( f"scrape_batch: failed to decode part for correlation_id={correlation_id!r}: {decode_err}", code="ERR::API::INTERNAL_ERROR", http_status_code=500, ) continue # API-generated error parts carry an error body instead of # the scrape envelope — surface them as typed per-part errors. if is_api_error_part(parsed, part_headers): try: part_error = error_from_api_error_part(parsed, part_headers, response.request) except Exception as factory_err: part_error = ScrapflyError( f"scrape_batch: malformed error part for correlation_id={correlation_id!r}: {factory_err}", code="ERR::API::INTERNAL_ERROR", http_status_code=500, ) yield correlation_id, part_error continue part_result = None try: api_response = ScrapeApiResponse( response=response, request=response.request, api_result=parsed, scrape_config=cfg, large_object_handler=self._handle_scrape_large_objects, ) # Don't auto-raise on upstream error — per-part errors # are surfaced via the yielded tuple, not exceptions. api_response.raise_for_result(raise_on_upstream_error=False) part_result = api_response except ScrapflyError as scrape_err: part_result = scrape_err except (EncoderError, Exception) as part_err: part_result = ScrapflyError( f"scrape_batch: failed to process part for correlation_id={correlation_id!r}: {part_err}", code="ERR::API::INTERNAL_ERROR", http_status_code=500, ) yield correlation_id, part_result finally: batch_session.close() def save_screenshot(self, screenshot_api_response:ScreenshotApiResponse, name:str, path:Optional[str]=None): """ Save a screenshot from a screenshot API response :param api_response: ScreenshotApiResponse :param name: str - name of the screenshot to save as :param path: Optional[str] """ if screenshot_api_response.screenshot_success is not True: raise RuntimeError('Screenshot was not successful') if not screenshot_api_response.image: raise RuntimeError('Screenshot binary does not exist') content = screenshot_api_response.image extension_name = screenshot_api_response.metadata['extension_name'] if path: os.makedirs(path, exist_ok=True) file_path = os.path.join(path, f'{name}.{extension_name}') else: file_path = f'{name}.{extension_name}' if isinstance(content, bytes): content = BytesIO(content) with open(file_path, 'wb') as f: shutil.copyfileobj(content, f, length=131072) def save_scrape_screenshot(self, api_response:ScrapeApiResponse, name:str, path:Optional[str]=None): """ Save a screenshot from a scrape result :param api_response: ScrapeApiResponse :param name: str - name of the screenshot given in the scrape config :param path: Optional[str] """ if not api_response.scrape_result['screenshots']: raise RuntimeError('Screenshot %s do no exists' % name) try: api_response.scrape_result['screenshots'][name] except KeyError: raise RuntimeError('Screenshot %s do no exists' % name) screenshot_response = self._http_handler( method='GET', url=api_response.scrape_result['screenshots'][name]['url'], params={'key': self.key}, verify=self.verify ) screenshot_response.raise_for_status() if not name.endswith('.jpg'): name += '.jpg' api_response.sink(path=path, name=name, content=screenshot_response.content) def sink(self, api_response:ScrapeApiResponse, content:Optional[Union[str, bytes]]=None, path: Optional[str] = None, name: Optional[str] = None, file: Optional[Union[TextIO, BytesIO]] = None) -> str: scrape_result = api_response.result['result'] scrape_config = api_response.result['config'] file_content = content or scrape_result['content'] file_path = None file_extension = None if name: name_parts = name.split('.') if len(name_parts) > 1: file_extension = name_parts[-1] if not file: if file_extension is None: try: mime_type = scrape_result['response_headers']['content-type'] except KeyError: mime_type = 'application/octet-stream' if ';' in mime_type: mime_type = mime_type.split(';')[0] file_extension = '.' + mime_type.split('/')[1] if not name: name = scrape_config['url'].split('/')[-1] if name.find(file_extension) == -1: name += file_extension file_path = path + '/' + name if path else name if file_path == file_extension: url = re.sub(r'(https|http)?://', '', api_response.config['url']).replace('/', '-') if url[-1] == '-': url = url[:-1] url += file_extension file_path = url file = open(file_path, 'wb') if isinstance(file_content, str): file_content = BytesIO(file_content.encode('utf-8')) elif isinstance(file_content, bytes): file_content = BytesIO(file_content) file_content.seek(0) with file as f: shutil.copyfileobj(file_content, f, length=131072) logger.info('file %s created' % file_path) return file_path def _handle_scrape_large_objects( self, callback_url:str, format: Literal['clob', 'blob'] ) -> Tuple[Union[BytesIO, str], str]: if format not in ['clob', 'blob']: raise ContentError('Large objects handle can handles format format [blob, clob], given: %s' % format) response = self._http_handler(**{ 'method': 'GET', 'url': callback_url, 'verify': self.verify, 'timeout': (self.connect_timeout, self.default_read_timeout), 'headers': { 'accept-encoding': self.body_handler.content_encoding, 'accept': self.body_handler.accept, 'user-agent': self.ua }, 'params': {'key': self.key} }) if self.body_handler.support(headers=response.headers): content = self.body_handler(content=response.content, content_type=response.headers['content-type']) else: content = response.content if format == 'clob': return content.decode('utf-8'), 'text' return BytesIO(content), 'binary' def _handle_api_response( self, response: Response, scrape_config:ScrapeConfig, raise_on_upstream_error: Optional[bool] = True ) -> ScrapeApiResponse: if scrape_config.method == 'HEAD': body = None else: if self.body_handler.support(headers=response.headers): body = self.body_handler(content=response.content, content_type=response.headers['content-type']) else: # body_handler rejected — content-type not in SUPPORTED_CONTENT_TYPES. # Response may still be compressed (zstd/brotli) if requests did # not transparently decompress. Probe content-encoding and try # the handler's read() anyway before falling back to a tolerant # utf-8 decode. Previously this branch raised UnicodeDecodeError # on valid zstd/br responses with a non-json/msgpack content-type. raw = response.content content_encoding = response.headers.get('content-encoding', '').lower() if content_encoding in ('gzip', 'gz', 'deflate', 'br', 'brotli', 'zstd'): try: raw = self.body_handler.read( content=raw, content_encoding=content_encoding, content_type=response.headers.get('content-type', ''), signature=None, ) except Exception: # Fall through to tolerant decode below; don't mask the # real error with a decoder crash. pass if isinstance(raw, (bytes, bytearray)): body = raw.decode('utf-8', errors='replace') else: body = raw api_response:ScrapeApiResponse = ScrapeApiResponse( response=response, request=response.request, api_result=body, scrape_config=scrape_config, large_object_handler=self._handle_scrape_large_objects ) api_response.raise_for_result(raise_on_upstream_error=raise_on_upstream_error) return api_response def _handle_screenshot_api_response( self, response: Response, screenshot_config:ScreenshotConfig, raise_on_upstream_error: Optional[bool] = True ) -> ScreenshotApiResponse: if self.body_handler.support(headers=response.headers): body = self.body_handler(content=response.content, content_type=response.headers['content-type']) else: body = {'result': response.content} api_response:ScreenshotApiResponse = ScreenshotApiResponse( response=response, request=response.request, api_result=body, screenshot_config=screenshot_config ) api_response.raise_for_result(raise_on_upstream_error=raise_on_upstream_error) return api_response def _handle_extraction_api_response( self, response: Response, extraction_config:ExtractionConfig, raise_on_upstream_error: Optional[bool] = True ) -> ExtractionApiResponse: if self.body_handler.support(headers=response.headers): body = self.body_handler(content=response.content, content_type=response.headers['content-type']) else: body = response.content.decode('utf-8') api_response:ExtractionApiResponse = ExtractionApiResponse( response=response, request=response.request, api_result=body, extraction_config=extraction_config ) api_response.raise_for_result(raise_on_upstream_error=raise_on_upstream_error) return api_response @backoff.on_exception(backoff.expo, exception=ConnectionError, max_tries=5) def start_crawl(self, crawler_config: CrawlerConfig) -> CrawlerStartResponse: """ Start a crawler job :param crawler_config: CrawlerConfig :return: CrawlerStartResponse with UUID and initial status Example: ```python from scrapfly import ScrapflyClient, CrawlerConfig client = ScrapflyClient(key='YOUR_API_KEY') config = CrawlerConfig( url='https://example.com', page_limit=100, max_depth=3 ) response = client.start_crawl(config) print(f"Crawler started: {response.uuid}") ``` """ # POST /crawl accepts two body formats: # - application/json: the entire crawler configuration as JSON. # Used for seed-URL crawls and remote_url_list crawls. # - multipart/form-data: a 'config' JSON part and a 'urls' text part # (one URL per line). Used only when the caller provides an # in-memory url_list, so we can stream it as a file payload # instead of inlining it into the JSON body. parts = crawler_config.to_multipart_parts() urls_blob = parts['urls'] query_params = {'key': self.key} timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) url = f'{self.host}/crawl' logger.debug(f"Crawler API POST {url}?key=***") if urls_blob is not None: config_body = json.dumps(parts['config']).encode('utf-8') files = { 'config': ('config.json', config_body, 'application/json'), 'urls': ('urls.txt', urls_blob.encode('utf-8'), 'text/plain'), } logger.debug( f"Crawler API multipart config: {parts['config']} ; " f"urls part: {len(urls_blob.splitlines())} URL(s)" ) response = self._http_handler( method='POST', url=url, params=query_params, files=files, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) else: logger.debug(f"Crawler API body: {parts['config']}") response = self._http_handler( method='POST', url=url, params=query_params, json=parts['config'], timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code not in (200, 201): # Log error details for debugging try: error_detail = response.json() except (ValueError, Exception): error_detail = response.text logger.debug(f"Crawler API error ({response.status_code}): {error_detail}") self._handle_crawler_error_response(response) result = response.json() return CrawlerStartResponse(result) @backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def get_crawl_status(self, uuid: str) -> CrawlerStatusResponse: """ Get crawler job status :param uuid: Crawler job UUID :return: CrawlerStatusResponse with progress information Example: ```python status = client.get_crawl_status(uuid) print(f"Status: {status.status}") print(f"Progress: {status.progress_pct:.1f}%") print(f"Crawled: {status.urls_crawled}/{status.urls_discovered}") if status.is_complete: print("Crawl completed!") ``` """ timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) response = self._http_handler( method='GET', url=f'{self.host}/crawl/{uuid}/status', params={'key': self.key}, # key as query param (already correct) timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response) result = response.json() return CrawlerStatusResponse(result) def cancel_crawl(self, crawl_uuid: str) -> bool: """ Cancel a running crawler job :param crawl_uuid: Crawler job UUID to cancel :return: True if cancelled successfully Example: ```python # Start a crawl crawl = client.start_crawl(config) # Cancel it client.cancel_crawl(crawl.uuid) ``` """ timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) response = self._http_handler( method='DELETE', url=f'{self.host}/crawl/{crawl_uuid}', params={'key': self.key}, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code not in (200, 204): self._handle_crawler_error_response(response) return True @backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def get_crawl_artifact( self, uuid: str, artifact_type: str = 'warc' ) -> CrawlerArtifactResponse: """ Download crawler job artifact :param uuid: Crawler job UUID :param artifact_type: Artifact type ('warc' or 'har') :return: CrawlerArtifactResponse with WARC data and parsing utilities Example: ```python # Wait for crawl to complete while True: status = client.get_crawl_status(uuid) if status.is_complete: break time.sleep(5) # Download artifact artifact = client.get_crawl_artifact(uuid) # Easy mode: get all pages pages = artifact.get_pages() for page in pages: print(f"{page['url']}: {page['status_code']}") # Memory-efficient: iterate for record in artifact.iter_responses(): process(record.content) # Save to file artifact.save('crawl.warc.gz') ``` """ timeout = (self.connect_timeout, 300) # 5 minutes for large downloads response = self._http_handler( method='GET', url=f'{self.host}/crawl/{uuid}/artifact', params={ 'key': self.key, 'type': artifact_type }, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response) return CrawlerArtifactResponse(response.content, artifact_type=artifact_type) @backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def get_crawl_contents( self, uuid: str, format: Literal['html', 'clean_html', 'markdown', 'json', 'text', 'extracted_data', 'page_metadata'] = 'html' ) -> Dict[str, Any]: """ Get crawl contents in a specific format Retrieves extracted content from crawled pages in the format(s) specified in your crawl configuration (via content_formats parameter). :param uuid: Crawler job UUID :param format: Content format - 'html', 'clean_html', 'markdown', 'json', 'text', 'extracted_data', 'page_metadata' :return: Dictionary with format {"contents": {url: content, ...}, "links": {...}} Example: ```python # Get all content in markdown format result = client.get_crawl_contents(uuid, format='markdown') contents = result['contents'] # Access specific URL for url, content in contents.items(): print(f"{url}: {len(content)} chars") ``` """ timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) params = { 'key': self.key, 'format': format } response = self._http_handler( method='GET', url=f'{self.host}/crawl/{uuid}/contents', params=params, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response) return response.json() @backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def get_crawl_urls( self, uuid: str, status: Optional[Literal['visited', 'pending', 'failed', 'skipped']] = None, page: int = 1, per_page: int = 100 ) -> CrawlerUrlsResponse: """ List the URLs of a crawler job ``GET /crawl/{uuid}/urls`` answers ``text/plain``, one record per line: the URL alone for 'visited' / 'pending', ``url,reason`` for 'failed' / 'skipped'. JSON is not offered on the success path, the endpoint being sized for millions of records per job. ``page`` and ``per_page`` are sent for parity with the other SDKs and echoed on the response, but the API forwards only the status filter to the crawler, so one call answers with the whole server-side page. :param uuid: Crawler job UUID :param status: URL status filter - 'visited', 'pending', 'failed', 'skipped'. None leaves the server default ('visited'). :param page: 1-based page number :param per_page: Page size :return: CrawlerUrlsResponse with the parsed entries and the echoed pagination Example: ```python urls = client.get_crawl_urls(uuid, status='failed') for entry in urls: print(f"{entry.url}: {entry.reason}") ``` """ timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) params = { 'key': self.key, 'page': page, 'per_page': per_page } if status is not None: params['status'] = status response = self._http_handler( method='GET', url=f'{self.host}/crawl/{uuid}/urls', params=params, timeout=timeout, headers={ 'User-Agent': self.ua, # text/plain is the success format; error envelopes come back # as JSON whatever the endpoint renders when it succeeds. 'Accept': 'text/plain, application/json' }, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response) # A JSON body on a 200 is an envelope the text parser would read as # records: every line of it becomes a bogus URL entry. Fail loud # instead of handing back a page of garbage. if 'application/json' in response.headers.get('Content-Type', ''): raise ScrapflyCrawlerError( message=( f"Crawler API returned JSON on a 200 for GET /crawl/{uuid}/urls, " f"expected text/plain: {response.text[:500]}" ), code='ERR::CRAWLER::UNEXPECTED_RESPONSE_FORMAT', http_status_code=response.status_code ) return CrawlerUrlsResponse.from_text( body=response.text, status_hint=status or 'visited', page=page, per_page=per_page ) @backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def crawl_search( self, crawl_ids: List[str], query: str, limit: int = 10, mode: Literal['vector', 'fts', 'hybrid'] = 'hybrid', filters: Optional[Dict[str, Any]] = None, cursor: Optional[str] = None ) -> CrawlerSearchResponse: """ Search across the search indexes of one or more crawls. The collection form is the real endpoint: ``POST /crawl/search`` fans out over ``crawl_ids`` and merges one global ranking. Only crawls started with ``CrawlerConfig(search=True)`` whose index reached ``READY``/``PARTIAL`` contribute; the others come back in ``response.skipped`` with a reason and never fail the call. :param crawl_ids: Crawler job UUIDs to search. Duplicates are rejected by the API. :param query: Free-text query. :param limit: Maximum results, 1-50 (server cap). :param mode: 'vector' (semantic), 'fts' (keyword) or 'hybrid' (both, merged with reciprocal rank fusion). :param filters: Optional flat filter map: 'url_prefix', 'host', 'source_format', 'content_type', 'http_status', 'crawler_uuid'. Unknown keys are rejected server-side. :param cursor: Opaque token from a previous response to fetch the next page. Paging is cursor-based; an offset over a partial fan-out would re-run the legs and shift ranks. :return: CrawlerSearchResponse Example: ```python results = client.crawl_search( crawl_ids=[uuid_a, uuid_b], query='TLS fingerprint', limit=20, ) for hit in results: print(f"{hit.rank}. {hit.url} ({hit.score:.3f})") ``` """ if not crawl_ids: raise ValueError("crawl_ids must contain at least one crawler UUID") if not query: raise ValueError("query cannot be empty") body: Dict[str, Any] = { 'query': query, 'crawl_ids': list(crawl_ids), 'limit': limit, 'mode': mode, } if filters: body['filters'] = filters if cursor: body['cursor'] = cursor timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_SEARCH_API_READ_TIMEOUT) response = self._http_handler( method='POST', url=f'{self.host}/crawl/search', params={'key': self.key}, json=body, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response, error_class=CrawlerSearchError) return CrawlerSearchResponse(response.json()) def crawl_prompt( self, crawl_ids: List[str], prompt: str, search: Optional[Dict[str, Any]] = None, model: Optional[str] = None, stream: bool = True ) -> Union[Iterator[CrawlerPromptEvent], Dict[str, Any]]: """ Ask a question answered from the content of one or more crawls. ``POST /crawl/prompt`` retrieves from the same fan-out as :py:meth:`crawl_search`, then generates an answer over the retrieved chunks. With ``stream=True`` (default) this returns an iterator of :class:`CrawlerPromptEvent`: ``source`` frames first, then ``token`` frames, then one ``done`` frame. The HTTP response stays open for the whole generation, so consume the iterator promptly and close it (or exhaust it) to release the connection. With ``stream=False`` the same content is returned as a single dict. No backoff decorator here: a retry would re-run the fan-out and the generation, and both are billable. :param crawl_ids: Crawler job UUIDs to answer from. :param prompt: The question. :param search: Optional retrieval overrides: 'limit', 'mode', 'filters'. Same grammar as :py:meth:`crawl_search`. :param model: Optional Gemini model id. Unset uses the server default. :param stream: Consume the answer as SSE frames (True) or as one JSON object (False). :return: Iterator[CrawlerPromptEvent] when streaming, else Dict Example: ```python for event in client.crawl_prompt([uuid], 'Summarize the pricing page'): if event.is_token: print(event.data, end='', flush=True) ``` """ if not crawl_ids: raise ValueError("crawl_ids must contain at least one crawler UUID") if not prompt: raise ValueError("prompt cannot be empty") body: Dict[str, Any] = { 'prompt': prompt, 'crawl_ids': list(crawl_ids), 'generation': {'stream': stream}, } if search: body['search'] = search if model: body['generation']['model'] = model timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_PROMPT_API_READ_TIMEOUT) response = self._http_handler( method='POST', url=f'{self.host}/crawl/prompt', params={'key': self.key}, json=body, timeout=timeout, headers={ 'User-Agent': self.ua, 'Accept': 'text/event-stream' if stream else 'application/json', }, stream=stream, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response, error_class=CrawlerPromptError) if not stream: return response.json() return self._iter_prompt_events(response) @staticmethod def _iter_prompt_events(response: Response) -> Iterator[CrawlerPromptEvent]: """ Decode the ``/crawl/prompt`` SSE body into typed frames. Only ``event:`` and ``data:`` are handled; ``:keepalive`` comment frames exist to keep intermediaries from closing an idle connection and carry nothing for the caller. ``token`` data is a JSON string; every other frame is a JSON object. A complete ``done`` frame terminates the iterator. EOF before that frame is an error, even if some tokens have already been delivered. Lines are decoded as UTF-8 rather than through ``iter_lines(decode_unicode=True)``: requests derives the encoding from the Content-Type and falls back to ISO-8859-1 for any ``text/*`` without a charset, which mangles every non-ASCII token. SSE is UTF-8 by specification. """ try: event_name: Optional[str] = None data_lines: List[str] = [] for raw in response.iter_lines(): if raw is None: continue line = raw.decode('utf-8', errors='replace').rstrip('\r') if line.startswith(':'): continue if line == '': # Blank line terminates a frame. if event_name is not None and data_lines: payload = '\n'.join(data_lines) try: data = json.loads(payload) except ValueError: data = payload if event_name == 'error': code = data.get('code', 'ERR::CRAWLER::UNKNOWN') if isinstance(data, dict) else 'ERR::CRAWLER::UNKNOWN' message = data.get('message', payload) if isinstance(data, dict) else payload raise CrawlerPromptError( message=message, code=code, http_status_code=response.status_code ) yield CrawlerPromptEvent(event=event_name, data=data) if event_name == 'done': return event_name = None data_lines = [] continue if line.startswith('event:'): event_name = line[len('event:'):].strip() elif line.startswith('data:'): data_lines.append(line[len('data:'):].lstrip(' ')) raise CrawlerPromptError( message='Prompt stream ended before the done frame', code='ERR::CRAWLER::PROMPT_GENERATION_FAILED', http_status_code=response.status_code ) finally: response.close() def crawl_refresh_now(self, uuid: str) -> CrawlerRefreshState: """ Run one refresh of an existing crawl immediately, without waiting for the next scheduled period. ``POST /crawl/{uuid}/refresh`` re-scrapes the crawl's own URLs in place: same ``crawler_uuid``, same artifacts, same search index. Only pages whose content actually changed are re-indexed, and pages that disappeared are dropped. The call returns as soon as the run is accepted; poll :py:meth:`get_crawl_status` or :py:meth:`crawl_refresh_history` for the outcome. A refresh bills the pages it re-scrapes, exactly like the original crawl. Pages whose fingerprint is unchanged still cost their scrape; what they save is the embedding and the index write. No backoff decorator here: a retry would start a second re-scrape of the whole site, and that is billable. :param uuid: Crawler job UUID. :return: CrawlerRefreshState Example: ```python state = client.crawl_refresh_now(uuid) print(state.status, state.generation) ``` """ if not uuid: raise ValueError("uuid must be a non-empty string") timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) response = self._http_handler( method='POST', url=f'{self.host}/crawl/{uuid}/refresh', params={'key': self.key}, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code not in (200, 202): self._handle_crawler_error_response(response, error_class=CrawlerRefreshError) return CrawlerRefreshState(response.json()) @backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def crawl_refresh_settings( self, uuid: str, enabled: Optional[bool] = None, interval_seconds: Optional[int] = None ) -> CrawlerRefreshState: """ Change the refresh schedule of an existing crawl. ``PATCH /crawl/{uuid}/refresh``. Both arguments are optional and only what is passed is changed, so turning a crawl off keeps its interval for when it is turned back on. Turning refresh on for a crawl that was started without it is allowed: the crawl already holds the URL index a refresh walks. :param uuid: Crawler job UUID. :param enabled: Turn auto-refresh on or off. :param interval_seconds: Period between runs, 3600 to 7776000 (1 hour to 90 days). :return: CrawlerRefreshState Example: ```python client.crawl_refresh_settings(uuid, enabled=True, interval_seconds=86400) ``` """ if not uuid: raise ValueError("uuid must be a non-empty string") if enabled is None and interval_seconds is None: raise ValueError("pass at least one of enabled, interval_seconds") if interval_seconds is not None and not (CrawlerConfig.REFRESH_MIN_INTERVAL <= interval_seconds <= CrawlerConfig.REFRESH_MAX_INTERVAL): raise ValueError( f"interval_seconds must be between {CrawlerConfig.REFRESH_MIN_INTERVAL} " f"and {CrawlerConfig.REFRESH_MAX_INTERVAL} seconds" ) # Wire keys are the ones POST /crawl already takes, so a crawl body and # a later PATCH name the same things. The `enabled` / `interval_seconds` # spelling belongs to the state block this call answers with, not to # its request; the API decodes the body with unknown fields rejected. body: Dict[str, Any] = {} if enabled is not None: body['refresh'] = enabled if interval_seconds is not None: body['refresh_interval'] = interval_seconds timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) response = self._http_handler( method='PATCH', url=f'{self.host}/crawl/{uuid}/refresh', params={'key': self.key}, json=body, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response, error_class=CrawlerRefreshError) return CrawlerRefreshState(response.json()) @backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def crawl_refresh_history(self, uuid: str, limit: Optional[int] = None) -> List[CrawlerRefreshEntry]: """ Read a crawl's refresh timeline, newest last. ``GET /crawl/{uuid}/refresh/history``. The server keeps the 50 most recent runs; older rows are trimmed rather than paged, because the timeline exists to show recent activity. :param uuid: Crawler job UUID. :param limit: Keep only the last N rows. :return: List[CrawlerRefreshEntry] Example: ```python for entry in client.crawl_refresh_history(uuid): print(entry.at, entry.updated, 'changed' if entry.changed else 'no change') ``` """ if not uuid: raise ValueError("uuid must be a non-empty string") params: Dict[str, Any] = {'key': self.key} if limit is not None: params['limit'] = limit timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) response = self._http_handler( method='GET', url=f'{self.host}/crawl/{uuid}/refresh/history', params=params, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response, error_class=CrawlerRefreshError) return CrawlerRefreshState(response.json()).history def _handle_crawler_error_response(self, response: Response, error_class: Optional[type] = None): """ Handle error responses from Crawler API. :param error_class: Optional CrawlerError subclass to raise instead of HttpError. Endpoint-specific codes (ERR::CRAWLER::SEARCH_*) are worth catching on their own, and HttpError cannot be narrowed. """ try: error_data = response.json() error_msg = error_data.get('message', 'Unknown error') error_code = error_data.get('code', 'ERR::CRAWLER::UNKNOWN') except Exception: error_msg = response.text error_code = 'ERR::CRAWLER::UNKNOWN' message = f"Crawler API error ({response.status_code}): {error_msg}" if error_class is not None: raise error_class( message=message, code=error_code, http_status_code=response.status_code ) raise HttpError( message=message, code=error_code, http_status_code=response.status_code, request=response.request, response=response ) def cloud_browser(self, browser_config: Optional[BrowserConfig] = None) -> str: """ Get the WebSocket URL for a Cloud Browser session. :param browser_config: Optional BrowserConfig - connection parameters :return: str - the full wss:// URL for CDP connection On rejection, the server sends a JSON error frame followed by a close frame with code 1008/1011/1013 and a "ERR::BROWSER::CODE: reason" string. See the docs for read patterns: https://scrapfly.io/docs/cloud-browser-api/errors#websocket-close-frame """ if browser_config is None: browser_config = BrowserConfig() return browser_config.websocket_url(api_key=self.key, host=self.cloud_browser_host) def cloud_browser_project_salt(self) -> str: """Return the deterministic project salt for this client's api_key. Matches the X-Browser-Project-Salt response header returned on VNC-enabled Cloud Browser upgrades, where the salt is also the VNC password prefix. Useful for verifying that an attach link belongs to your project before sharing it. """ return BrowserConfig.project_salt(self.key) def cloud_browser_vnc_password(self, browser_config: BrowserConfig) -> str: """Return the password a native VNC client must type to attach to a session created with this config: "<project_salt>-<vnc_password>". Copy this value into your VNC client when connecting to the TCP endpoint (port 5901). The WebSocket endpoint /run/<run_id>/vnc takes the raw vnc_password instead. """ return browser_config.vnc_client_password(self.key) def cloud_browser_unblock( self, url: str, country: Optional[str] = None, os: Optional[str] = None, browser_brand: Optional[str] = None, session: Optional[str] = None, timeout: Optional[int] = None, browser_timeout: Optional[int] = None, enable_mcp: Optional[bool] = None, debug: Optional[bool] = None, ) -> Dict: """ Bypass anti-bot protection and get a ready-to-use browser session. Unblock always uses a residential proxy (no proxy pool selection) and always performs a GET — the endpoint's purpose is to harvest cookies / clearance tokens via an ASP-bypass navigation, not to proxy arbitrary HTTP requests. :param url: Target URL to navigate to and bypass protection :param country: ISO country code for residential proxy geolocation :param os: Operating system fingerprint: 'linux', 'windows', 'macos', 'android', 'iphone', 'ipad' :param browser_brand: Browser brand fingerprint: 'chrome', 'edge', 'brave', 'opera' :param session: Named session for reconnection — reuses the existing ASP session when one exists and disables auto-close on disconnect :param timeout: Navigation timeout in seconds (max 300) :param browser_timeout: Browser session timeout in seconds (max 1800) :param enable_mcp: Enable MCP streamable-HTTP endpoint on the session :param debug: When True, the session is recorded and accessible via cloud_browser_playback / cloud_browser_video using the returned run_id :return: dict with ws_url, session_id, run_id """ json_body = {'url': url} if country is not None: json_body['country'] = country if os is not None: json_body['os'] = os if browser_brand is not None: json_body['browser_brand'] = browser_brand if session is not None: json_body['session'] = session if timeout is not None: json_body['timeout'] = timeout if browser_timeout is not None: json_body['browser_timeout'] = browser_timeout if enable_mcp is not None: json_body['enable_mcp'] = enable_mcp if debug is not None: json_body['debug'] = debug response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/unblock', json=json_body, params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, 155), headers={ 'content-type': 'application/json', 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_session_stop(self, session_id: str) -> None: """ Terminate a Cloud Browser session. :param session_id: The session identifier to terminate """ response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/session/' + session_id + '/stop', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() def cloud_browser_playback(self, run_id: str) -> Dict: """ Get playback info for a debug session recording. :param run_id: The unique run identifier :return: dict with available, status, metadata, video_url, retry_after_ms """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/run/' + run_id + '/playback', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_wait_for_playback( self, run_id: str, timeout: float = 180.0, poll_interval_fallback: float = 3.0, ) -> Dict: """ Poll the playback endpoint until the recording resolves to a terminal state (status='ready' or status='unavailable') or the timeout elapses. Honours the server-side retry_after_ms hint whenever it is present. :param run_id: The unique run identifier :param timeout: Maximum seconds to wait for the recording to be ready :param poll_interval_fallback: Delay (s) used when the server does not return a retry_after_ms hint :return: Final playback dict — the same shape as cloud_browser_playback """ import time deadline = time.monotonic() + timeout while True: playback = self.cloud_browser_playback(run_id) status = playback.get('status') if status != 'uploading': return playback remaining = deadline - time.monotonic() if remaining <= 0: return playback retry_after_ms = playback.get('retry_after_ms') or int(poll_interval_fallback * 1000) sleep_for = min(retry_after_ms / 1000.0, remaining) time.sleep(sleep_for) def cloud_browser_video(self, run_id: str, save_path: Optional[str] = None) -> bytes: """ Download a debug session recording video. :param run_id: The unique run identifier :param save_path: Optional file path to save the video (e.g. 'recording.webm') :return: bytes - raw video data """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/run/' + run_id + '/video', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, 120), # Videos can be large headers={ 'user-agent': self.ua }, stream=True, ) response.raise_for_status() data = response.content if save_path: with open(save_path, 'wb') as f: f.write(data) return data # --- Cloud Browser Extension Management --- def cloud_browser_extension_list(self) -> Dict: """ List all browser extensions for the current account. :return: dict with 'extensions' list and 'quota' info (used, limit) """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/extension', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_extension_get(self, extension_id: str) -> Dict: """ Get details of a specific browser extension. :param extension_id: The extension identifier :return: dict with extension details """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/extension/' + extension_id, params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_extension_upload(self, file_path: str) -> Dict: """ Upload a browser extension from a local file (.zip or .crx). :param file_path: Path to the extension file :return: dict with 'extension' details and 'is_update' flag """ with open(file_path, 'rb') as f: response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/extension', params={'key': self.key}, files={'file': (os.path.basename(file_path), f)}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_extension_upload_from_url(self, extension_url: str) -> Dict: """ Install a browser extension from a URL pointing to a .crx file. URL-based extensions auto-update on each browser session start. :param extension_url: URL to the .crx extension file :return: dict with 'extension' details and 'is_update' flag """ response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/extension', params={'key': self.key}, json={'extension_url': extension_url}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'content-type': 'application/json', 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_extension_delete(self, extension_id: str) -> Dict: """ Delete a browser extension. :param extension_id: The extension identifier to delete :return: dict with success status """ response = self._http_handler( method='DELETE', url=self.cloud_browser_api_host + '/extension/' + extension_id, params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_sessions(self) -> Dict: """ List all running Cloud Browser sessions. :return: dict with 'sessions' list and 'total' count """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/sessions', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json() # --- Cloud Browser Credential Vault --- # # End-to-end encrypted credential storage for Cloud Browser sessions. # The vault key is generated server-side at create + rotate time and # returned in the response body exactly once. The server does NOT # persist the key — clients must save it locally on receipt. # # SECURITY: NEVER log, print, or include vault_key in exception # messages, debug output, repr, or any breadcrumb. The whole product # property is "Scrapfly receives the key transiently and zeros it." # Any leak invalidates that guarantee. def cloud_browser_vault_create(self, name: str, description: Optional[str] = None) -> Dict: """ Create a new credential vault. The response includes the freshly generated vault key under the `key` field — this is the ONLY time the server returns it. Save it immediately; it cannot be recovered. :param name: Human-readable vault name :param description: Optional description :return: dict with `vault`, `key`, and `message` """ body: Dict[str, str] = {'name': name} if description is not None: body['description'] = description response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/vault', params={'key': self.key}, json=body, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'content-type': 'application/json', 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_vault_list(self) -> Dict: """ List all credential vaults on the account (no secret material). :return: dict with `vaults` list """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/vault', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_vault_get(self, vault_id: str) -> Dict: """ Fetch metadata for a single vault (no secret material). :param vault_id: The vault identifier :return: dict with `vault` envelope """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/vault/' + vault_id, params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_vault_update( self, vault_id: str, name: Optional[str] = None, description: Optional[str] = None, ) -> Dict: """ Update vault metadata (name and/or description). Does NOT touch encrypted material; X-Vault-Key is not required. :param vault_id: The vault identifier :param name: Optional new name :param description: Optional new description :return: server response dict """ body: Dict[str, str] = {} if name is not None: body['name'] = name if description is not None: body['description'] = description response = self._http_handler( method='PATCH', url=self.cloud_browser_api_host + '/vault/' + vault_id, params={'key': self.key}, json=body, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'content-type': 'application/json', 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_vault_delete(self, vault_id: str) -> Dict: """ Delete a vault and all its items. :param vault_id: The vault identifier :return: server response dict """ response = self._http_handler( method='DELETE', url=self.cloud_browser_api_host + '/vault/' + vault_id, params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_vault_rotate(self, vault_id: str, current_vault_key: str) -> Dict: """ Rotate the vault encryption key. Requires the CURRENT key in the X-Vault-Key header. Server generates a fresh key, rewraps every item, and returns the new key in the response body exactly once. After this call, the old key cannot read any row in the vault. :param vault_id: The vault identifier :param current_vault_key: The current base64-encoded vault key (forwarded as X-Vault-Key, never logged) :return: dict with `key` and `message` """ response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/vault/' + vault_id + '/rotate', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua, 'X-Vault-Key': current_vault_key, }, ) response.raise_for_status() return response.json() def cloud_browser_vault_item_list(self, vault_id: str) -> Dict: """ List items in a vault (metadata only — no secret material). :param vault_id: The vault identifier :return: dict with `items` list """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/vault/' + vault_id + '/item', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json() def cloud_browser_vault_item_create( self, vault_id: str, vault_key: str, type: str, label: str, origin: str, secret: Dict, username: Optional[str] = None, ) -> Dict: """ Add an item to a vault. Requires the vault key in X-Vault-Key — the server uses it to wrap a per-row DEK that encrypts the secret. :param vault_id: The vault identifier :param vault_key: The base64-encoded vault key (X-Vault-Key, never logged) :param type: Item type ("password", "passkey", "cookie", "totp") :param label: Human-readable label :param origin: Origin URL the credential is bound to :param secret: Typed secret payload, e.g. ``{"password": "hunter2"}`` :param username: Optional username :return: dict with `item` and `message` """ body: Dict = { 'type': type, 'label': label, 'origin': origin, 'secret': secret, } if username is not None: body['username'] = username response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/vault/' + vault_id + '/item', params={'key': self.key}, json=body, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'content-type': 'application/json', 'user-agent': self.ua, 'X-Vault-Key': vault_key, }, ) response.raise_for_status() return response.json() def cloud_browser_vault_item_update( self, vault_id: str, item_id: str, vault_key: Optional[str] = None, label: Optional[str] = None, origin: Optional[str] = None, username: Optional[str] = None, secret: Optional[Dict] = None, type: Optional[str] = None, ) -> Dict: """ Update an item. Metadata-only patches (label/origin/username) do NOT require X-Vault-Key. Patching the secret triggers re-encryption and REQUIRES both vault_key AND type (the server's parseSecret() switches on type to route the typed payload). :param vault_id: The vault identifier :param item_id: The item identifier :param vault_key: Required iff `secret` is not None. Forwarded as X-Vault-Key; never logged. :param label: Optional new label :param origin: Optional new origin :param username: Optional new username :param secret: Optional new typed secret payload (rotates the encrypted blob on the server) :param type: Item type ("password", "passkey", "cookie", "totp"), required iff `secret` is not None. :return: server response dict """ if secret is not None and not vault_key: raise ValueError( "vault_key is required when secret is provided " "(server requires X-Vault-Key for re-encryption)" ) if secret is not None and not type: raise ValueError( "type is required when secret is provided " "(server's parseSecret switches on it)" ) body: Dict = {} if label is not None: body['label'] = label if origin is not None: body['origin'] = origin if username is not None: body['username'] = username if secret is not None: body['secret'] = secret body['type'] = type headers = { 'content-type': 'application/json', 'user-agent': self.ua, } if vault_key is not None: headers['X-Vault-Key'] = vault_key response = self._http_handler( method='PATCH', url=self.cloud_browser_api_host + '/vault/' + vault_id + '/item/' + item_id, params={'key': self.key}, json=body, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers=headers, ) response.raise_for_status() return response.json() def cloud_browser_vault_item_delete(self, vault_id: str, item_id: str) -> Dict: """ Delete an item from a vault. :param vault_id: The vault identifier :param item_id: The item identifier :return: server response dict """ response = self._http_handler( method='DELETE', url=self.cloud_browser_api_host + '/vault/' + vault_id + '/item/' + item_id, params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json()Mixed into ScrapflyClient — provides the public schedule surface.
All methods funnel through
_schedule_request, which uses the sameself._http_handlerandself.host/self.keyas the rest of the client so retries, verify, headers and timeouts behave identically.Ancestors
Class variables
var CLOUD_BROWSER_API_HOSTvar CLOUD_BROWSER_HOSTvar CONCURRENCY_AUTOvar DATETIME_FORMATvar DEFAULT_CONNECT_TIMEOUTvar DEFAULT_CRAWLER_API_READ_TIMEOUTvar DEFAULT_CRAWLER_PROMPT_API_READ_TIMEOUTvar DEFAULT_CRAWLER_SEARCH_API_READ_TIMEOUTvar DEFAULT_EXTRACTION_API_READ_TIMEOUTvar DEFAULT_READ_TIMEOUTvar DEFAULT_SCREENSHOT_API_READ_TIMEOUTvar DEFAULT_WEBSCRAPING_API_READ_TIMEOUTvar HOSTvar brotli : boolvar connect_timeout : intvar debug : boolvar default_read_timeout : intvar distributed_mode : boolvar extraction_api_read_timeout : intvar host : strvar key : strvar max_concurrency : intvar monitoring_api_read_timeout : intvar read_timeout : intvar reporter : scrapfly.reporter.Reportervar screenshot_api_read_timeout : intvar verify : boolvar version : strvar web_scraping_api_read_timeout : int
Instance variables
prop http-
Expand source code
@property def http(self): return self._http_handler prop ua : str-
Expand source code
@property def ua(self) -> str: return 'ScrapflySDK/%s (Python %s, %s, %s)' % ( self.version, platform.python_version(), platform.uname().system, platform.uname().machine )
Methods
def account(self) ‑> str | Dict-
Expand source code
def account(self) -> Union[str, Dict]: response = self._http_handler( method='GET', url=self.host + '/account', params={'key': self.key}, verify=self.verify, headers={ 'accept-encoding': self.body_handler.content_encoding, 'accept': self.body_handler.accept, 'user-agent': self.ua }, ) response.raise_for_status() if self.body_handler.support(response.headers): return self.body_handler(response.content, response.headers['content-type']) return response.content.decode('utf-8') async def async_extraction(self,
extraction_config: ExtractionConfig,
loop: asyncio.events.AbstractEventLoop | None = None) ‑> ExtractionApiResponse-
Expand source code
async def async_extraction(self, extraction_config:ExtractionConfig, loop:Optional[AbstractEventLoop]=None) -> ExtractionApiResponse: if loop is None: loop = asyncio.get_running_loop() return await loop.run_in_executor(self.async_executor, self.extract, extraction_config) async def async_scrape(self,
scrape_config: ScrapeConfig,
loop: asyncio.events.AbstractEventLoop | None = None) ‑> ScrapeApiResponse-
Expand source code
async def async_scrape(self, scrape_config:ScrapeConfig, loop:Optional[AbstractEventLoop]=None) -> ScrapeApiResponse: if loop is None: loop = asyncio.get_running_loop() return await loop.run_in_executor(self.async_executor, self.scrape, scrape_config) async def async_screenshot(self,
screenshot_config: ScreenshotConfig,
loop: asyncio.events.AbstractEventLoop | None = None) ‑> ScreenshotApiResponse-
Expand source code
async def async_screenshot(self, screenshot_config:ScreenshotConfig, loop:Optional[AbstractEventLoop]=None) -> ScreenshotApiResponse: if loop is None: loop = asyncio.get_running_loop() return await loop.run_in_executor(self.async_executor, self.screenshot, screenshot_config) def cancel_crawl(self, crawl_uuid: str) ‑> bool-
Expand source code
def cancel_crawl(self, crawl_uuid: str) -> bool: """ Cancel a running crawler job :param crawl_uuid: Crawler job UUID to cancel :return: True if cancelled successfully Example: ```python # Start a crawl crawl = client.start_crawl(config) # Cancel it client.cancel_crawl(crawl.uuid) ``` """ timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) response = self._http_handler( method='DELETE', url=f'{self.host}/crawl/{crawl_uuid}', params={'key': self.key}, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code not in (200, 204): self._handle_crawler_error_response(response) return TrueCancel a running crawler job
:param crawl_uuid: Crawler job UUID to cancel :return: True if cancelled successfully
Example
# Start a crawl crawl = client.start_crawl(config) # Cancel it client.cancel_crawl(crawl.uuid) def classify(self,
url: str,
status_code: int,
headers: Dict[str, str] | None = None,
body: str | None = None,
method: str = 'GET') ‑> ClassifyResult-
Expand source code
def classify( self, url: str, status_code: int, headers: Optional[Dict[str, str]] = None, body: Optional[str] = None, method: str = "GET", ) -> ClassifyResult: """Classify an already-fetched HTTP response for anti-bot blocking. Runs the same 80+ shield pipeline used by live Scrapfly scrapes against a response you already have (from your own proxy, cache, etc). 1 API credit per call. See https://scrapfly.io/docs/scrape-api/classify for the full contract. """ if not url: raise ContentError("classify: url is required") if not (100 <= int(status_code) <= 599): raise ContentError( "classify: status_code must be a valid HTTP status in [100, 599]" ) payload: Dict[str, Any] = { "url": url, "status_code": int(status_code), "method": method or "GET", } if headers: payload["headers"] = {str(k): str(v) for k, v in headers.items()} if body is not None: payload["body"] = body response = self._http_handler( method="POST", url=self.host + "/classify", params={"key": self.key}, json=payload, verify=self.verify, headers={ "accept-encoding": self.body_handler.content_encoding, "accept": self.body_handler.accept, "user-agent": self.ua, "content-type": "application/json", }, ) response.raise_for_status() if self.body_handler.support(response.headers): data = self.body_handler(response.content, response.headers["content-type"]) else: import json as _json data = _json.loads(response.content.decode("utf-8")) return ClassifyResult.from_dict(data)Classify an already-fetched HTTP response for anti-bot blocking.
Runs the same 80+ shield pipeline used by live Scrapfly scrapes against a response you already have (from your own proxy, cache, etc). 1 API credit per call. See https://scrapfly.io/docs/scrape-api/classify for the full contract.
def close(self)-
Expand source code
def close(self): if self.http_session is not None: self.http_session.close() self.http_session = None # The executor is created in __init__ and owns worker threads that # outlive the HTTP session; shutting it down here prevents thread # leaks for callers that reuse the client across open()/close() # cycles or rely on GC to reclaim it. if self.async_executor is not None: self.async_executor.shutdown(wait=False) self.async_executor = None def cloud_browser(self,
browser_config: BrowserConfig | None = None) ‑> str-
Expand source code
def cloud_browser(self, browser_config: Optional[BrowserConfig] = None) -> str: """ Get the WebSocket URL for a Cloud Browser session. :param browser_config: Optional BrowserConfig - connection parameters :return: str - the full wss:// URL for CDP connection On rejection, the server sends a JSON error frame followed by a close frame with code 1008/1011/1013 and a "ERR::BROWSER::CODE: reason" string. See the docs for read patterns: https://scrapfly.io/docs/cloud-browser-api/errors#websocket-close-frame """ if browser_config is None: browser_config = BrowserConfig() return browser_config.websocket_url(api_key=self.key, host=self.cloud_browser_host)Get the WebSocket URL for a Cloud Browser session.
:param browser_config: Optional BrowserConfig - connection parameters :return: str - the full wss:// URL for CDP connection
On rejection, the server sends a JSON error frame followed by a close frame with code 1008/1011/1013 and a "ERR::BROWSER::CODE: reason" string. See the docs for read patterns: https://scrapfly.io/docs/cloud-browser-api/errors#websocket-close-frame
def cloud_browser_extension_delete(self, extension_id: str) ‑> Dict-
Expand source code
def cloud_browser_extension_delete(self, extension_id: str) -> Dict: """ Delete a browser extension. :param extension_id: The extension identifier to delete :return: dict with success status """ response = self._http_handler( method='DELETE', url=self.cloud_browser_api_host + '/extension/' + extension_id, params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json()Delete a browser extension. :param extension_id: The extension identifier to delete :return: dict with success status
def cloud_browser_extension_get(self, extension_id: str) ‑> Dict-
Expand source code
def cloud_browser_extension_get(self, extension_id: str) -> Dict: """ Get details of a specific browser extension. :param extension_id: The extension identifier :return: dict with extension details """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/extension/' + extension_id, params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json()Get details of a specific browser extension. :param extension_id: The extension identifier :return: dict with extension details
def cloud_browser_extension_list(self) ‑> Dict-
Expand source code
def cloud_browser_extension_list(self) -> Dict: """ List all browser extensions for the current account. :return: dict with 'extensions' list and 'quota' info (used, limit) """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/extension', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json()List all browser extensions for the current account. :return: dict with 'extensions' list and 'quota' info (used, limit)
def cloud_browser_extension_upload(self, file_path: str) ‑> Dict-
Expand source code
def cloud_browser_extension_upload(self, file_path: str) -> Dict: """ Upload a browser extension from a local file (.zip or .crx). :param file_path: Path to the extension file :return: dict with 'extension' details and 'is_update' flag """ with open(file_path, 'rb') as f: response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/extension', params={'key': self.key}, files={'file': (os.path.basename(file_path), f)}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json()Upload a browser extension from a local file (.zip or .crx). :param file_path: Path to the extension file :return: dict with 'extension' details and 'is_update' flag
def cloud_browser_extension_upload_from_url(self, extension_url: str) ‑> Dict-
Expand source code
def cloud_browser_extension_upload_from_url(self, extension_url: str) -> Dict: """ Install a browser extension from a URL pointing to a .crx file. URL-based extensions auto-update on each browser session start. :param extension_url: URL to the .crx extension file :return: dict with 'extension' details and 'is_update' flag """ response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/extension', params={'key': self.key}, json={'extension_url': extension_url}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'content-type': 'application/json', 'user-agent': self.ua }, ) response.raise_for_status() return response.json()Install a browser extension from a URL pointing to a .crx file. URL-based extensions auto-update on each browser session start. :param extension_url: URL to the .crx extension file :return: dict with 'extension' details and 'is_update' flag
def cloud_browser_playback(self, run_id: str) ‑> Dict-
Expand source code
def cloud_browser_playback(self, run_id: str) -> Dict: """ Get playback info for a debug session recording. :param run_id: The unique run identifier :return: dict with available, status, metadata, video_url, retry_after_ms """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/run/' + run_id + '/playback', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json()Get playback info for a debug session recording. :param run_id: The unique run identifier :return: dict with available, status, metadata, video_url, retry_after_ms
def cloud_browser_project_salt(self) ‑> str-
Expand source code
def cloud_browser_project_salt(self) -> str: """Return the deterministic project salt for this client's api_key. Matches the X-Browser-Project-Salt response header returned on VNC-enabled Cloud Browser upgrades, where the salt is also the VNC password prefix. Useful for verifying that an attach link belongs to your project before sharing it. """ return BrowserConfig.project_salt(self.key)Return the deterministic project salt for this client's api_key. Matches the X-Browser-Project-Salt response header returned on VNC-enabled Cloud Browser upgrades, where the salt is also the VNC password prefix. Useful for verifying that an attach link belongs to your project before sharing it.
def cloud_browser_session_stop(self, session_id: str) ‑> None-
Expand source code
def cloud_browser_session_stop(self, session_id: str) -> None: """ Terminate a Cloud Browser session. :param session_id: The session identifier to terminate """ response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/session/' + session_id + '/stop', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status()Terminate a Cloud Browser session. :param session_id: The session identifier to terminate
def cloud_browser_sessions(self) ‑> Dict-
Expand source code
def cloud_browser_sessions(self) -> Dict: """ List all running Cloud Browser sessions. :return: dict with 'sessions' list and 'total' count """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/sessions', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json()List all running Cloud Browser sessions. :return: dict with 'sessions' list and 'total' count
def cloud_browser_unblock(self,
url: str,
country: str | None = None,
os: str | None = None,
browser_brand: str | None = None,
session: str | None = None,
timeout: int | None = None,
browser_timeout: int | None = None,
enable_mcp: bool | None = None,
debug: bool | None = None) ‑> Dict-
Expand source code
def cloud_browser_unblock( self, url: str, country: Optional[str] = None, os: Optional[str] = None, browser_brand: Optional[str] = None, session: Optional[str] = None, timeout: Optional[int] = None, browser_timeout: Optional[int] = None, enable_mcp: Optional[bool] = None, debug: Optional[bool] = None, ) -> Dict: """ Bypass anti-bot protection and get a ready-to-use browser session. Unblock always uses a residential proxy (no proxy pool selection) and always performs a GET — the endpoint's purpose is to harvest cookies / clearance tokens via an ASP-bypass navigation, not to proxy arbitrary HTTP requests. :param url: Target URL to navigate to and bypass protection :param country: ISO country code for residential proxy geolocation :param os: Operating system fingerprint: 'linux', 'windows', 'macos', 'android', 'iphone', 'ipad' :param browser_brand: Browser brand fingerprint: 'chrome', 'edge', 'brave', 'opera' :param session: Named session for reconnection — reuses the existing ASP session when one exists and disables auto-close on disconnect :param timeout: Navigation timeout in seconds (max 300) :param browser_timeout: Browser session timeout in seconds (max 1800) :param enable_mcp: Enable MCP streamable-HTTP endpoint on the session :param debug: When True, the session is recorded and accessible via cloud_browser_playback / cloud_browser_video using the returned run_id :return: dict with ws_url, session_id, run_id """ json_body = {'url': url} if country is not None: json_body['country'] = country if os is not None: json_body['os'] = os if browser_brand is not None: json_body['browser_brand'] = browser_brand if session is not None: json_body['session'] = session if timeout is not None: json_body['timeout'] = timeout if browser_timeout is not None: json_body['browser_timeout'] = browser_timeout if enable_mcp is not None: json_body['enable_mcp'] = enable_mcp if debug is not None: json_body['debug'] = debug response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/unblock', json=json_body, params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, 155), headers={ 'content-type': 'application/json', 'user-agent': self.ua }, ) response.raise_for_status() return response.json()Bypass anti-bot protection and get a ready-to-use browser session.
Unblock always uses a residential proxy (no proxy pool selection) and always performs a GET — the endpoint's purpose is to harvest cookies / clearance tokens via an ASP-bypass navigation, not to proxy arbitrary HTTP requests.
:param url: Target URL to navigate to and bypass protection :param country: ISO country code for residential proxy geolocation :param os: Operating system fingerprint: 'linux', 'windows', 'macos', 'android', 'iphone', 'ipad' :param browser_brand: Browser brand fingerprint: 'chrome', 'edge', 'brave', 'opera' :param session: Named session for reconnection — reuses the existing ASP session when one exists and disables auto-close on disconnect :param timeout: Navigation timeout in seconds (max 300) :param browser_timeout: Browser session timeout in seconds (max 1800) :param enable_mcp: Enable MCP streamable-HTTP endpoint on the session :param debug: When True, the session is recorded and accessible via cloud_browser_playback / cloud_browser_video using the returned run_id :return: dict with ws_url, session_id, run_id
def cloud_browser_vault_create(self, name: str, description: str | None = None) ‑> Dict-
Expand source code
def cloud_browser_vault_create(self, name: str, description: Optional[str] = None) -> Dict: """ Create a new credential vault. The response includes the freshly generated vault key under the `key` field — this is the ONLY time the server returns it. Save it immediately; it cannot be recovered. :param name: Human-readable vault name :param description: Optional description :return: dict with `vault`, `key`, and `message` """ body: Dict[str, str] = {'name': name} if description is not None: body['description'] = description response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/vault', params={'key': self.key}, json=body, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'content-type': 'application/json', 'user-agent': self.ua }, ) response.raise_for_status() return response.json()Create a new credential vault. The response includes the freshly generated vault key under the
keyfield — this is the ONLY time the server returns it. Save it immediately; it cannot be recovered.:param name: Human-readable vault name :param description: Optional description :return: dict with
vault,key, andmessage def cloud_browser_vault_delete(self, vault_id: str) ‑> Dict-
Expand source code
def cloud_browser_vault_delete(self, vault_id: str) -> Dict: """ Delete a vault and all its items. :param vault_id: The vault identifier :return: server response dict """ response = self._http_handler( method='DELETE', url=self.cloud_browser_api_host + '/vault/' + vault_id, params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json()Delete a vault and all its items. :param vault_id: The vault identifier :return: server response dict
def cloud_browser_vault_get(self, vault_id: str) ‑> Dict-
Expand source code
def cloud_browser_vault_get(self, vault_id: str) -> Dict: """ Fetch metadata for a single vault (no secret material). :param vault_id: The vault identifier :return: dict with `vault` envelope """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/vault/' + vault_id, params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json()Fetch metadata for a single vault (no secret material). :param vault_id: The vault identifier :return: dict with
vaultenvelope def cloud_browser_vault_item_create(self,
vault_id: str,
vault_key: str,
type: str,
label: str,
origin: str,
secret: Dict,
username: str | None = None) ‑> Dict-
Expand source code
def cloud_browser_vault_item_create( self, vault_id: str, vault_key: str, type: str, label: str, origin: str, secret: Dict, username: Optional[str] = None, ) -> Dict: """ Add an item to a vault. Requires the vault key in X-Vault-Key — the server uses it to wrap a per-row DEK that encrypts the secret. :param vault_id: The vault identifier :param vault_key: The base64-encoded vault key (X-Vault-Key, never logged) :param type: Item type ("password", "passkey", "cookie", "totp") :param label: Human-readable label :param origin: Origin URL the credential is bound to :param secret: Typed secret payload, e.g. ``{"password": "hunter2"}`` :param username: Optional username :return: dict with `item` and `message` """ body: Dict = { 'type': type, 'label': label, 'origin': origin, 'secret': secret, } if username is not None: body['username'] = username response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/vault/' + vault_id + '/item', params={'key': self.key}, json=body, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'content-type': 'application/json', 'user-agent': self.ua, 'X-Vault-Key': vault_key, }, ) response.raise_for_status() return response.json()Add an item to a vault. Requires the vault key in X-Vault-Key — the server uses it to wrap a per-row DEK that encrypts the secret.
:param vault_id: The vault identifier :param vault_key: The base64-encoded vault key (X-Vault-Key, never logged) :param type: Item type ("password", "passkey", "cookie", "totp") :param label: Human-readable label :param origin: Origin URL the credential is bound to :param secret: Typed secret payload, e.g.
{"password": "hunter2"}:param username: Optional username :return: dict withitemandmessage def cloud_browser_vault_item_delete(self, vault_id: str, item_id: str) ‑> Dict-
Expand source code
def cloud_browser_vault_item_delete(self, vault_id: str, item_id: str) -> Dict: """ Delete an item from a vault. :param vault_id: The vault identifier :param item_id: The item identifier :return: server response dict """ response = self._http_handler( method='DELETE', url=self.cloud_browser_api_host + '/vault/' + vault_id + '/item/' + item_id, params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json()Delete an item from a vault. :param vault_id: The vault identifier :param item_id: The item identifier :return: server response dict
def cloud_browser_vault_item_list(self, vault_id: str) ‑> Dict-
Expand source code
def cloud_browser_vault_item_list(self, vault_id: str) -> Dict: """ List items in a vault (metadata only — no secret material). :param vault_id: The vault identifier :return: dict with `items` list """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/vault/' + vault_id + '/item', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json()List items in a vault (metadata only — no secret material). :param vault_id: The vault identifier :return: dict with
itemslist def cloud_browser_vault_item_update(self,
vault_id: str,
item_id: str,
vault_key: str | None = None,
label: str | None = None,
origin: str | None = None,
username: str | None = None,
secret: Dict | None = None,
type: str | None = None) ‑> Dict-
Expand source code
def cloud_browser_vault_item_update( self, vault_id: str, item_id: str, vault_key: Optional[str] = None, label: Optional[str] = None, origin: Optional[str] = None, username: Optional[str] = None, secret: Optional[Dict] = None, type: Optional[str] = None, ) -> Dict: """ Update an item. Metadata-only patches (label/origin/username) do NOT require X-Vault-Key. Patching the secret triggers re-encryption and REQUIRES both vault_key AND type (the server's parseSecret() switches on type to route the typed payload). :param vault_id: The vault identifier :param item_id: The item identifier :param vault_key: Required iff `secret` is not None. Forwarded as X-Vault-Key; never logged. :param label: Optional new label :param origin: Optional new origin :param username: Optional new username :param secret: Optional new typed secret payload (rotates the encrypted blob on the server) :param type: Item type ("password", "passkey", "cookie", "totp"), required iff `secret` is not None. :return: server response dict """ if secret is not None and not vault_key: raise ValueError( "vault_key is required when secret is provided " "(server requires X-Vault-Key for re-encryption)" ) if secret is not None and not type: raise ValueError( "type is required when secret is provided " "(server's parseSecret switches on it)" ) body: Dict = {} if label is not None: body['label'] = label if origin is not None: body['origin'] = origin if username is not None: body['username'] = username if secret is not None: body['secret'] = secret body['type'] = type headers = { 'content-type': 'application/json', 'user-agent': self.ua, } if vault_key is not None: headers['X-Vault-Key'] = vault_key response = self._http_handler( method='PATCH', url=self.cloud_browser_api_host + '/vault/' + vault_id + '/item/' + item_id, params={'key': self.key}, json=body, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers=headers, ) response.raise_for_status() return response.json()Update an item. Metadata-only patches (label/origin/username) do NOT require X-Vault-Key. Patching the secret triggers re-encryption and REQUIRES both vault_key AND type (the server's parseSecret() switches on type to route the typed payload).
:param vault_id: The vault identifier :param item_id: The item identifier :param vault_key: Required iff
secretis not None. Forwarded as X-Vault-Key; never logged. :param label: Optional new label :param origin: Optional new origin :param username: Optional new username :param secret: Optional new typed secret payload (rotates the encrypted blob on the server) :param type: Item type ("password", "passkey", "cookie", "totp"), required iffsecretis not None. :return: server response dict def cloud_browser_vault_list(self) ‑> Dict-
Expand source code
def cloud_browser_vault_list(self) -> Dict: """ List all credential vaults on the account (no secret material). :return: dict with `vaults` list """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/vault', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua }, ) response.raise_for_status() return response.json()List all credential vaults on the account (no secret material). :return: dict with
vaultslist def cloud_browser_vault_rotate(self, vault_id: str, current_vault_key: str) ‑> Dict-
Expand source code
def cloud_browser_vault_rotate(self, vault_id: str, current_vault_key: str) -> Dict: """ Rotate the vault encryption key. Requires the CURRENT key in the X-Vault-Key header. Server generates a fresh key, rewraps every item, and returns the new key in the response body exactly once. After this call, the old key cannot read any row in the vault. :param vault_id: The vault identifier :param current_vault_key: The current base64-encoded vault key (forwarded as X-Vault-Key, never logged) :return: dict with `key` and `message` """ response = self._http_handler( method='POST', url=self.cloud_browser_api_host + '/vault/' + vault_id + '/rotate', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'user-agent': self.ua, 'X-Vault-Key': current_vault_key, }, ) response.raise_for_status() return response.json()Rotate the vault encryption key. Requires the CURRENT key in the X-Vault-Key header. Server generates a fresh key, rewraps every item, and returns the new key in the response body exactly once. After this call, the old key cannot read any row in the vault.
:param vault_id: The vault identifier :param current_vault_key: The current base64-encoded vault key (forwarded as X-Vault-Key, never logged) :return: dict with
keyandmessage def cloud_browser_vault_update(self, vault_id: str, name: str | None = None, description: str | None = None) ‑> Dict-
Expand source code
def cloud_browser_vault_update( self, vault_id: str, name: Optional[str] = None, description: Optional[str] = None, ) -> Dict: """ Update vault metadata (name and/or description). Does NOT touch encrypted material; X-Vault-Key is not required. :param vault_id: The vault identifier :param name: Optional new name :param description: Optional new description :return: server response dict """ body: Dict[str, str] = {} if name is not None: body['name'] = name if description is not None: body['description'] = description response = self._http_handler( method='PATCH', url=self.cloud_browser_api_host + '/vault/' + vault_id, params={'key': self.key}, json=body, verify=self.verify, timeout=(self.connect_timeout, self.default_read_timeout), headers={ 'content-type': 'application/json', 'user-agent': self.ua }, ) response.raise_for_status() return response.json()Update vault metadata (name and/or description). Does NOT touch encrypted material; X-Vault-Key is not required.
:param vault_id: The vault identifier :param name: Optional new name :param description: Optional new description :return: server response dict
def cloud_browser_video(self, run_id: str, save_path: str | None = None) ‑> bytes-
Expand source code
def cloud_browser_video(self, run_id: str, save_path: Optional[str] = None) -> bytes: """ Download a debug session recording video. :param run_id: The unique run identifier :param save_path: Optional file path to save the video (e.g. 'recording.webm') :return: bytes - raw video data """ response = self._http_handler( method='GET', url=self.cloud_browser_api_host + '/run/' + run_id + '/video', params={'key': self.key}, verify=self.verify, timeout=(self.connect_timeout, 120), # Videos can be large headers={ 'user-agent': self.ua }, stream=True, ) response.raise_for_status() data = response.content if save_path: with open(save_path, 'wb') as f: f.write(data) return dataDownload a debug session recording video. :param run_id: The unique run identifier :param save_path: Optional file path to save the video (e.g. 'recording.webm') :return: bytes - raw video data
def cloud_browser_vnc_password(self,
browser_config: BrowserConfig) ‑> str-
Expand source code
def cloud_browser_vnc_password(self, browser_config: BrowserConfig) -> str: """Return the password a native VNC client must type to attach to a session created with this config: "<project_salt>-<vnc_password>". Copy this value into your VNC client when connecting to the TCP endpoint (port 5901). The WebSocket endpoint /run/<run_id>/vnc takes the raw vnc_password instead. """ return browser_config.vnc_client_password(self.key)Return the password a native VNC client must type to attach to a session created with this config: "
- ". Copy this value into your VNC client when connecting to the TCP endpoint (port 5901). The WebSocket endpoint /run/
/vnc takes the raw vnc_password instead. def cloud_browser_wait_for_playback(self, run_id: str, timeout: float = 180.0, poll_interval_fallback: float = 3.0) ‑> Dict-
Expand source code
def cloud_browser_wait_for_playback( self, run_id: str, timeout: float = 180.0, poll_interval_fallback: float = 3.0, ) -> Dict: """ Poll the playback endpoint until the recording resolves to a terminal state (status='ready' or status='unavailable') or the timeout elapses. Honours the server-side retry_after_ms hint whenever it is present. :param run_id: The unique run identifier :param timeout: Maximum seconds to wait for the recording to be ready :param poll_interval_fallback: Delay (s) used when the server does not return a retry_after_ms hint :return: Final playback dict — the same shape as cloud_browser_playback """ import time deadline = time.monotonic() + timeout while True: playback = self.cloud_browser_playback(run_id) status = playback.get('status') if status != 'uploading': return playback remaining = deadline - time.monotonic() if remaining <= 0: return playback retry_after_ms = playback.get('retry_after_ms') or int(poll_interval_fallback * 1000) sleep_for = min(retry_after_ms / 1000.0, remaining) time.sleep(sleep_for)Poll the playback endpoint until the recording resolves to a terminal state (status='ready' or status='unavailable') or the timeout elapses. Honours the server-side retry_after_ms hint whenever it is present.
:param run_id: The unique run identifier :param timeout: Maximum seconds to wait for the recording to be ready :param poll_interval_fallback: Delay (s) used when the server does not return a retry_after_ms hint :return: Final playback dict — the same shape as cloud_browser_playback
async def concurrent_scrape(self,
scrape_configs: List[ScrapeConfig],
concurrency: int | None = None)-
Expand source code
async def concurrent_scrape(self, scrape_configs:List[ScrapeConfig], concurrency:Optional[int]=None): if concurrency is None: concurrency = self.max_concurrency elif concurrency == self.CONCURRENCY_AUTO: concurrency = self.account()['subscription']['max_concurrency'] loop = asyncio.get_running_loop() processing_tasks = [] results = [] processed_tasks = 0 expected_tasks = len(scrape_configs) def scrape_done_callback(task:Task): nonlocal processed_tasks try: if task.cancelled() is True: return error = task.exception() if error is not None: results.append(error) else: results.append(task.result()) finally: processing_tasks.remove(task) processed_tasks += 1 while scrape_configs or results or processing_tasks: logger.info("Scrape %d/%d - %d running" % (processed_tasks, expected_tasks, len(processing_tasks))) if scrape_configs: if len(processing_tasks) < concurrency: # @todo handle backpressure for _ in range(0, concurrency - len(processing_tasks)): try: scrape_config = scrape_configs.pop() except IndexError: break scrape_config.raise_on_upstream_error = False task = loop.create_task(self.async_scrape(scrape_config=scrape_config, loop=loop)) processing_tasks.append(task) task.add_done_callback(scrape_done_callback) for _ in results: result = results.pop() yield result await asyncio.sleep(.5) logger.debug("Scrape %d/%d - %d running" % (processed_tasks, expected_tasks, len(processing_tasks))) def crawl_prompt(self,
crawl_ids: List[str],
prompt: str,
search: Dict[str, Any] | None = None,
model: str | None = None,
stream: bool = True) ‑> Iterator[CrawlerPromptEvent] | Dict[str, Any]-
Expand source code
def crawl_prompt( self, crawl_ids: List[str], prompt: str, search: Optional[Dict[str, Any]] = None, model: Optional[str] = None, stream: bool = True ) -> Union[Iterator[CrawlerPromptEvent], Dict[str, Any]]: """ Ask a question answered from the content of one or more crawls. ``POST /crawl/prompt`` retrieves from the same fan-out as :py:meth:`crawl_search`, then generates an answer over the retrieved chunks. With ``stream=True`` (default) this returns an iterator of :class:`CrawlerPromptEvent`: ``source`` frames first, then ``token`` frames, then one ``done`` frame. The HTTP response stays open for the whole generation, so consume the iterator promptly and close it (or exhaust it) to release the connection. With ``stream=False`` the same content is returned as a single dict. No backoff decorator here: a retry would re-run the fan-out and the generation, and both are billable. :param crawl_ids: Crawler job UUIDs to answer from. :param prompt: The question. :param search: Optional retrieval overrides: 'limit', 'mode', 'filters'. Same grammar as :py:meth:`crawl_search`. :param model: Optional Gemini model id. Unset uses the server default. :param stream: Consume the answer as SSE frames (True) or as one JSON object (False). :return: Iterator[CrawlerPromptEvent] when streaming, else Dict Example: ```python for event in client.crawl_prompt([uuid], 'Summarize the pricing page'): if event.is_token: print(event.data, end='', flush=True) ``` """ if not crawl_ids: raise ValueError("crawl_ids must contain at least one crawler UUID") if not prompt: raise ValueError("prompt cannot be empty") body: Dict[str, Any] = { 'prompt': prompt, 'crawl_ids': list(crawl_ids), 'generation': {'stream': stream}, } if search: body['search'] = search if model: body['generation']['model'] = model timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_PROMPT_API_READ_TIMEOUT) response = self._http_handler( method='POST', url=f'{self.host}/crawl/prompt', params={'key': self.key}, json=body, timeout=timeout, headers={ 'User-Agent': self.ua, 'Accept': 'text/event-stream' if stream else 'application/json', }, stream=stream, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response, error_class=CrawlerPromptError) if not stream: return response.json() return self._iter_prompt_events(response)Ask a question answered from the content of one or more crawls.
POST /crawl/promptretrieves from the same fan-out as :py:meth:crawl_search, then generates an answer over the retrieved chunks.With
stream=True(default) this returns an iterator of :class:CrawlerPromptEvent:sourceframes first, thentokenframes, then onedoneframe. The HTTP response stays open for the whole generation, so consume the iterator promptly and close it (or exhaust it) to release the connection. Withstream=Falsethe same content is returned as a single dict.No backoff decorator here: a retry would re-run the fan-out and the generation, and both are billable.
:param crawl_ids: Crawler job UUIDs to answer from. :param prompt: The question. :param search: Optional retrieval overrides: 'limit', 'mode', 'filters'. Same grammar as :py:meth:
crawl_search. :param model: Optional Gemini model id. Unset uses the server default. :param stream: Consume the answer as SSE frames (True) or as one JSON object (False). :return: Iterator[CrawlerPromptEvent] when streaming, else DictExample
for event in client.crawl_prompt([uuid], 'Summarize the pricing page'): if event.is_token: print(event.data, end='', flush=True) def crawl_refresh_history(self, uuid: str, limit: int | None = None) ‑> List[CrawlerRefreshEntry]-
Expand source code
@backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def crawl_refresh_history(self, uuid: str, limit: Optional[int] = None) -> List[CrawlerRefreshEntry]: """ Read a crawl's refresh timeline, newest last. ``GET /crawl/{uuid}/refresh/history``. The server keeps the 50 most recent runs; older rows are trimmed rather than paged, because the timeline exists to show recent activity. :param uuid: Crawler job UUID. :param limit: Keep only the last N rows. :return: List[CrawlerRefreshEntry] Example: ```python for entry in client.crawl_refresh_history(uuid): print(entry.at, entry.updated, 'changed' if entry.changed else 'no change') ``` """ if not uuid: raise ValueError("uuid must be a non-empty string") params: Dict[str, Any] = {'key': self.key} if limit is not None: params['limit'] = limit timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) response = self._http_handler( method='GET', url=f'{self.host}/crawl/{uuid}/refresh/history', params=params, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response, error_class=CrawlerRefreshError) return CrawlerRefreshState(response.json()).historyRead a crawl's refresh timeline, newest last.
GET /crawl/{uuid}/refresh/history. The server keeps the 50 most recent runs; older rows are trimmed rather than paged, because the timeline exists to show recent activity.:param uuid: Crawler job UUID. :param limit: Keep only the last N rows. :return: List[CrawlerRefreshEntry]
Example
for entry in client.crawl_refresh_history(uuid): print(entry.at, entry.updated, 'changed' if entry.changed else 'no change') def crawl_refresh_now(self, uuid: str) ‑> CrawlerRefreshState-
Expand source code
def crawl_refresh_now(self, uuid: str) -> CrawlerRefreshState: """ Run one refresh of an existing crawl immediately, without waiting for the next scheduled period. ``POST /crawl/{uuid}/refresh`` re-scrapes the crawl's own URLs in place: same ``crawler_uuid``, same artifacts, same search index. Only pages whose content actually changed are re-indexed, and pages that disappeared are dropped. The call returns as soon as the run is accepted; poll :py:meth:`get_crawl_status` or :py:meth:`crawl_refresh_history` for the outcome. A refresh bills the pages it re-scrapes, exactly like the original crawl. Pages whose fingerprint is unchanged still cost their scrape; what they save is the embedding and the index write. No backoff decorator here: a retry would start a second re-scrape of the whole site, and that is billable. :param uuid: Crawler job UUID. :return: CrawlerRefreshState Example: ```python state = client.crawl_refresh_now(uuid) print(state.status, state.generation) ``` """ if not uuid: raise ValueError("uuid must be a non-empty string") timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) response = self._http_handler( method='POST', url=f'{self.host}/crawl/{uuid}/refresh', params={'key': self.key}, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code not in (200, 202): self._handle_crawler_error_response(response, error_class=CrawlerRefreshError) return CrawlerRefreshState(response.json())Run one refresh of an existing crawl immediately, without waiting for the next scheduled period.
POST /crawl/{uuid}/refreshre-scrapes the crawl's own URLs in place: samecrawler_uuid, same artifacts, same search index. Only pages whose content actually changed are re-indexed, and pages that disappeared are dropped. The call returns as soon as the run is accepted; poll :py:meth:get_crawl_statusor :py:meth:crawl_refresh_historyfor the outcome.A refresh bills the pages it re-scrapes, exactly like the original crawl. Pages whose fingerprint is unchanged still cost their scrape; what they save is the embedding and the index write.
No backoff decorator here: a retry would start a second re-scrape of the whole site, and that is billable.
:param uuid: Crawler job UUID. :return: CrawlerRefreshState
Example
state = client.crawl_refresh_now(uuid) print(state.status, state.generation) def crawl_refresh_settings(self, uuid: str, enabled: bool | None = None, interval_seconds: int | None = None) ‑> CrawlerRefreshState-
Expand source code
@backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def crawl_refresh_settings( self, uuid: str, enabled: Optional[bool] = None, interval_seconds: Optional[int] = None ) -> CrawlerRefreshState: """ Change the refresh schedule of an existing crawl. ``PATCH /crawl/{uuid}/refresh``. Both arguments are optional and only what is passed is changed, so turning a crawl off keeps its interval for when it is turned back on. Turning refresh on for a crawl that was started without it is allowed: the crawl already holds the URL index a refresh walks. :param uuid: Crawler job UUID. :param enabled: Turn auto-refresh on or off. :param interval_seconds: Period between runs, 3600 to 7776000 (1 hour to 90 days). :return: CrawlerRefreshState Example: ```python client.crawl_refresh_settings(uuid, enabled=True, interval_seconds=86400) ``` """ if not uuid: raise ValueError("uuid must be a non-empty string") if enabled is None and interval_seconds is None: raise ValueError("pass at least one of enabled, interval_seconds") if interval_seconds is not None and not (CrawlerConfig.REFRESH_MIN_INTERVAL <= interval_seconds <= CrawlerConfig.REFRESH_MAX_INTERVAL): raise ValueError( f"interval_seconds must be between {CrawlerConfig.REFRESH_MIN_INTERVAL} " f"and {CrawlerConfig.REFRESH_MAX_INTERVAL} seconds" ) # Wire keys are the ones POST /crawl already takes, so a crawl body and # a later PATCH name the same things. The `enabled` / `interval_seconds` # spelling belongs to the state block this call answers with, not to # its request; the API decodes the body with unknown fields rejected. body: Dict[str, Any] = {} if enabled is not None: body['refresh'] = enabled if interval_seconds is not None: body['refresh_interval'] = interval_seconds timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) response = self._http_handler( method='PATCH', url=f'{self.host}/crawl/{uuid}/refresh', params={'key': self.key}, json=body, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response, error_class=CrawlerRefreshError) return CrawlerRefreshState(response.json())Change the refresh schedule of an existing crawl.
PATCH /crawl/{uuid}/refresh. Both arguments are optional and only what is passed is changed, so turning a crawl off keeps its interval for when it is turned back on.Turning refresh on for a crawl that was started without it is allowed: the crawl already holds the URL index a refresh walks.
:param uuid: Crawler job UUID. :param enabled: Turn auto-refresh on or off. :param interval_seconds: Period between runs, 3600 to 7776000 (1 hour to 90 days). :return: CrawlerRefreshState
Example
client.crawl_refresh_settings(uuid, enabled=True, interval_seconds=86400) def crawl_search(self,
crawl_ids: List[str],
query: str,
limit: int = 10,
mode: Literal['vector', 'fts', 'hybrid'] = 'hybrid',
filters: Dict[str, Any] | None = None,
cursor: str | None = None) ‑> CrawlerSearchResponse-
Expand source code
@backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def crawl_search( self, crawl_ids: List[str], query: str, limit: int = 10, mode: Literal['vector', 'fts', 'hybrid'] = 'hybrid', filters: Optional[Dict[str, Any]] = None, cursor: Optional[str] = None ) -> CrawlerSearchResponse: """ Search across the search indexes of one or more crawls. The collection form is the real endpoint: ``POST /crawl/search`` fans out over ``crawl_ids`` and merges one global ranking. Only crawls started with ``CrawlerConfig(search=True)`` whose index reached ``READY``/``PARTIAL`` contribute; the others come back in ``response.skipped`` with a reason and never fail the call. :param crawl_ids: Crawler job UUIDs to search. Duplicates are rejected by the API. :param query: Free-text query. :param limit: Maximum results, 1-50 (server cap). :param mode: 'vector' (semantic), 'fts' (keyword) or 'hybrid' (both, merged with reciprocal rank fusion). :param filters: Optional flat filter map: 'url_prefix', 'host', 'source_format', 'content_type', 'http_status', 'crawler_uuid'. Unknown keys are rejected server-side. :param cursor: Opaque token from a previous response to fetch the next page. Paging is cursor-based; an offset over a partial fan-out would re-run the legs and shift ranks. :return: CrawlerSearchResponse Example: ```python results = client.crawl_search( crawl_ids=[uuid_a, uuid_b], query='TLS fingerprint', limit=20, ) for hit in results: print(f"{hit.rank}. {hit.url} ({hit.score:.3f})") ``` """ if not crawl_ids: raise ValueError("crawl_ids must contain at least one crawler UUID") if not query: raise ValueError("query cannot be empty") body: Dict[str, Any] = { 'query': query, 'crawl_ids': list(crawl_ids), 'limit': limit, 'mode': mode, } if filters: body['filters'] = filters if cursor: body['cursor'] = cursor timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_SEARCH_API_READ_TIMEOUT) response = self._http_handler( method='POST', url=f'{self.host}/crawl/search', params={'key': self.key}, json=body, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response, error_class=CrawlerSearchError) return CrawlerSearchResponse(response.json())Search across the search indexes of one or more crawls.
The collection form is the real endpoint:
POST /crawl/searchfans out overcrawl_idsand merges one global ranking. Only crawls started withCrawlerConfig(search=True)whose index reachedREADY/PARTIALcontribute; the others come back inresponse.skippedwith a reason and never fail the call.:param crawl_ids: Crawler job UUIDs to search. Duplicates are rejected by the API. :param query: Free-text query. :param limit: Maximum results, 1-50 (server cap). :param mode: 'vector' (semantic), 'fts' (keyword) or 'hybrid' (both, merged with reciprocal rank fusion). :param filters: Optional flat filter map: 'url_prefix', 'host', 'source_format', 'content_type', 'http_status', 'crawler_uuid'. Unknown keys are rejected server-side. :param cursor: Opaque token from a previous response to fetch the next page. Paging is cursor-based; an offset over a partial fan-out would re-run the legs and shift ranks. :return: CrawlerSearchResponse
Example
results = client.crawl_search( crawl_ids=[uuid_a, uuid_b], query='TLS fingerprint', limit=20, ) for hit in results: print(f"{hit.rank}. {hit.url} ({hit.score:.3f})") def extract(self,
extraction_config: ExtractionConfig,
no_raise: bool = False) ‑> ExtractionApiResponse-
Expand source code
@backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def extract(self, extraction_config:ExtractionConfig, no_raise:bool=False) -> ExtractionApiResponse: """ Extract structured data from text content :param extraction_config: ExtractionConfig :param no_raise: bool - if True, do not raise exception on error while the extraction api response is a ScrapflyError for seamless integration :return: str If you use no_raise=True, make sure to check the extraction_api_response.error attribute to handle the error. If the error is not none, you will get the following structure for example 'error': { 'code': 'ERR::EXTRACTION::CONTENT_TYPE_NOT_SUPPORTED', 'message': 'The content type of the response is not supported for extraction', 'http_code': 422, 'links': { 'Checkout the related doc: https://scrapfly.io/docs/extraction-api/error/ERR::EXTRACTION::CONTENT_TYPE_NOT_SUPPORTED' } } """ try: logger.debug('--> %s Extracting data from' % (extraction_config.content_type)) request_data = self._extraction_request(extraction_config=extraction_config) response = self._http_handler(**request_data) extraction_api_response = self._handle_extraction_response(response=response, extraction_config=extraction_config) return extraction_api_response except BaseException as e: self.reporter.report(error=e) if no_raise and isinstance(e, ScrapflyError) and e.api_response is not None: return e.api_response raise eExtract structured data from text content :param extraction_config: ExtractionConfig :param no_raise: bool - if True, do not raise exception on error while the extraction api response is a ScrapflyError for seamless integration :return: str
If you use no_raise=True, make sure to check the extraction_api_response.error attribute to handle the error. If the error is not none, you will get the following structure for example
'error': { 'code': 'ERR::EXTRACTION::CONTENT_TYPE_NOT_SUPPORTED', 'message': 'The content type of the response is not supported for extraction', 'http_code': 422, 'links': { 'Checkout the related doc: https://scrapfly.io/docs/extraction-api/error/ERR::EXTRACTION::CONTENT_TYPE_NOT_SUPPORTED' } }
def get_browser_monitoring_metrics(self,
period: str | None = None,
proxy_pool: str | None = None,
start: datetime.datetime | None = None,
end: datetime.datetime | None = None)-
Expand source code
def get_browser_monitoring_metrics( self, period:Optional[str]=None, proxy_pool:Optional[str]=None, start:Optional[datetime.datetime]=None, end:Optional[datetime.datetime]=None, ): if (start is not None and end is None) or (start is None and end is not None): raise ValueError('You must provide both start and end date') params:dict = {'key': self.key} if start is not None and end is not None: params['start'] = self._format_monitoring_dt(start) params['end'] = self._format_monitoring_dt(end) elif period is not None: params['period'] = period if proxy_pool is not None: params['proxy_pool'] = proxy_pool return self._monitoring_request('/browser/monitoring/metrics', params) def get_browser_monitoring_timeseries(self,
period: str | None = None,
proxy_pool: str | None = None,
start: datetime.datetime | None = None,
end: datetime.datetime | None = None)-
Expand source code
def get_browser_monitoring_timeseries( self, period:Optional[str]=None, proxy_pool:Optional[str]=None, start:Optional[datetime.datetime]=None, end:Optional[datetime.datetime]=None, ): if (start is not None and end is None) or (start is None and end is not None): raise ValueError('You must provide both start and end date') params:dict = {'key': self.key} if start is not None and end is not None: params['start'] = self._format_monitoring_dt(start) params['end'] = self._format_monitoring_dt(end) elif period is not None: params['period'] = period if proxy_pool is not None: params['proxy_pool'] = proxy_pool return self._monitoring_request('/browser/monitoring/metrics/timeseries', params) def get_crawl_artifact(self, uuid: str, artifact_type: str = 'warc') ‑> CrawlerArtifactResponse-
Expand source code
@backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def get_crawl_artifact( self, uuid: str, artifact_type: str = 'warc' ) -> CrawlerArtifactResponse: """ Download crawler job artifact :param uuid: Crawler job UUID :param artifact_type: Artifact type ('warc' or 'har') :return: CrawlerArtifactResponse with WARC data and parsing utilities Example: ```python # Wait for crawl to complete while True: status = client.get_crawl_status(uuid) if status.is_complete: break time.sleep(5) # Download artifact artifact = client.get_crawl_artifact(uuid) # Easy mode: get all pages pages = artifact.get_pages() for page in pages: print(f"{page['url']}: {page['status_code']}") # Memory-efficient: iterate for record in artifact.iter_responses(): process(record.content) # Save to file artifact.save('crawl.warc.gz') ``` """ timeout = (self.connect_timeout, 300) # 5 minutes for large downloads response = self._http_handler( method='GET', url=f'{self.host}/crawl/{uuid}/artifact', params={ 'key': self.key, 'type': artifact_type }, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response) return CrawlerArtifactResponse(response.content, artifact_type=artifact_type)Download crawler job artifact
:param uuid: Crawler job UUID :param artifact_type: Artifact type ('warc' or 'har') :return: CrawlerArtifactResponse with WARC data and parsing utilities
Example
# Wait for crawl to complete while True: status = client.get_crawl_status(uuid) if status.is_complete: break time.sleep(5) # Download artifact artifact = client.get_crawl_artifact(uuid) # Easy mode: get all pages pages = artifact.get_pages() for page in pages: print(f"{page['url']}: {page['status_code']}") # Memory-efficient: iterate for record in artifact.iter_responses(): process(record.content) # Save to file artifact.save('crawl.warc.gz') def get_crawl_contents(self,
uuid: str,
format: Literal['html', 'clean_html', 'markdown', 'json', 'text', 'extracted_data', 'page_metadata'] = 'html') ‑> Dict[str, Any]-
Expand source code
@backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def get_crawl_contents( self, uuid: str, format: Literal['html', 'clean_html', 'markdown', 'json', 'text', 'extracted_data', 'page_metadata'] = 'html' ) -> Dict[str, Any]: """ Get crawl contents in a specific format Retrieves extracted content from crawled pages in the format(s) specified in your crawl configuration (via content_formats parameter). :param uuid: Crawler job UUID :param format: Content format - 'html', 'clean_html', 'markdown', 'json', 'text', 'extracted_data', 'page_metadata' :return: Dictionary with format {"contents": {url: content, ...}, "links": {...}} Example: ```python # Get all content in markdown format result = client.get_crawl_contents(uuid, format='markdown') contents = result['contents'] # Access specific URL for url, content in contents.items(): print(f"{url}: {len(content)} chars") ``` """ timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) params = { 'key': self.key, 'format': format } response = self._http_handler( method='GET', url=f'{self.host}/crawl/{uuid}/contents', params=params, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response) return response.json()Get crawl contents in a specific format
Retrieves extracted content from crawled pages in the format(s) specified in your crawl configuration (via content_formats parameter).
:param uuid: Crawler job UUID :param format: Content format - 'html', 'clean_html', 'markdown', 'json', 'text', 'extracted_data', 'page_metadata' :return: Dictionary with format {"contents": {url: content, …}, "links": {…}}
Example
# Get all content in markdown format result = client.get_crawl_contents(uuid, format='markdown') contents = result['contents'] # Access specific URL for url, content in contents.items(): print(f"{url}: {len(content)} chars") def get_crawl_status(self, uuid: str) ‑> CrawlerStatusResponse-
Expand source code
@backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def get_crawl_status(self, uuid: str) -> CrawlerStatusResponse: """ Get crawler job status :param uuid: Crawler job UUID :return: CrawlerStatusResponse with progress information Example: ```python status = client.get_crawl_status(uuid) print(f"Status: {status.status}") print(f"Progress: {status.progress_pct:.1f}%") print(f"Crawled: {status.urls_crawled}/{status.urls_discovered}") if status.is_complete: print("Crawl completed!") ``` """ timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) response = self._http_handler( method='GET', url=f'{self.host}/crawl/{uuid}/status', params={'key': self.key}, # key as query param (already correct) timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response) result = response.json() return CrawlerStatusResponse(result)Get crawler job status
:param uuid: Crawler job UUID :return: CrawlerStatusResponse with progress information
Example
status = client.get_crawl_status(uuid) print(f"Status: {status.status}") print(f"Progress: {status.progress_pct:.1f}%") print(f"Crawled: {status.urls_crawled}/{status.urls_discovered}") if status.is_complete: print("Crawl completed!") def get_crawl_urls(self,
uuid: str,
status: Literal['visited', 'pending', 'failed', 'skipped'] | None = None,
page: int = 1,
per_page: int = 100) ‑> CrawlerUrlsResponse-
Expand source code
@backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def get_crawl_urls( self, uuid: str, status: Optional[Literal['visited', 'pending', 'failed', 'skipped']] = None, page: int = 1, per_page: int = 100 ) -> CrawlerUrlsResponse: """ List the URLs of a crawler job ``GET /crawl/{uuid}/urls`` answers ``text/plain``, one record per line: the URL alone for 'visited' / 'pending', ``url,reason`` for 'failed' / 'skipped'. JSON is not offered on the success path, the endpoint being sized for millions of records per job. ``page`` and ``per_page`` are sent for parity with the other SDKs and echoed on the response, but the API forwards only the status filter to the crawler, so one call answers with the whole server-side page. :param uuid: Crawler job UUID :param status: URL status filter - 'visited', 'pending', 'failed', 'skipped'. None leaves the server default ('visited'). :param page: 1-based page number :param per_page: Page size :return: CrawlerUrlsResponse with the parsed entries and the echoed pagination Example: ```python urls = client.get_crawl_urls(uuid, status='failed') for entry in urls: print(f"{entry.url}: {entry.reason}") ``` """ timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) params = { 'key': self.key, 'page': page, 'per_page': per_page } if status is not None: params['status'] = status response = self._http_handler( method='GET', url=f'{self.host}/crawl/{uuid}/urls', params=params, timeout=timeout, headers={ 'User-Agent': self.ua, # text/plain is the success format; error envelopes come back # as JSON whatever the endpoint renders when it succeeds. 'Accept': 'text/plain, application/json' }, verify=self.verify ) if response.status_code != 200: self._handle_crawler_error_response(response) # A JSON body on a 200 is an envelope the text parser would read as # records: every line of it becomes a bogus URL entry. Fail loud # instead of handing back a page of garbage. if 'application/json' in response.headers.get('Content-Type', ''): raise ScrapflyCrawlerError( message=( f"Crawler API returned JSON on a 200 for GET /crawl/{uuid}/urls, " f"expected text/plain: {response.text[:500]}" ), code='ERR::CRAWLER::UNEXPECTED_RESPONSE_FORMAT', http_status_code=response.status_code ) return CrawlerUrlsResponse.from_text( body=response.text, status_hint=status or 'visited', page=page, per_page=per_page )List the URLs of a crawler job
GET /crawl/{uuid}/urlsanswerstext/plain, one record per line: the URL alone for 'visited' / 'pending',url,reasonfor 'failed' / 'skipped'. JSON is not offered on the success path, the endpoint being sized for millions of records per job.pageandper_pageare sent for parity with the other SDKs and echoed on the response, but the API forwards only the status filter to the crawler, so one call answers with the whole server-side page.:param uuid: Crawler job UUID :param status: URL status filter - 'visited', 'pending', 'failed', 'skipped'. None leaves the server default ('visited'). :param page: 1-based page number :param per_page: Page size :return: CrawlerUrlsResponse with the parsed entries and the echoed pagination
Example
urls = client.get_crawl_urls(uuid, status='failed') for entry in urls: print(f"{entry.url}: {entry.reason}") def get_crawler_monitoring_metrics(self,
format: str = 'structured',
period: str | None = None,
aggregation: List[Literal['account', 'project', 'target']] | None = None,
include_webhook: bool = False)-
Expand source code
def get_crawler_monitoring_metrics( self, format:str=ScraperAPI.MONITORING_DATA_FORMAT_STRUCTURED, period:Optional[str]=None, aggregation:Optional[List[MonitoringAggregation]]=None, include_webhook:bool=False, ): return self._monitoring_request( '/crawl/monitoring/metrics', self._build_metrics_params(format, period, aggregation, include_webhook), ) def get_crawler_monitoring_target_metrics(self,
domain: str,
group_subdomain: bool = False,
period: Literal['subscription', 'last7d', 'last24h', 'last1h', 'last5m'] | None = 'last24h',
start: datetime.datetime | None = None,
end: datetime.datetime | None = None,
include_webhook: bool = False)-
Expand source code
def get_crawler_monitoring_target_metrics( self, domain:str, group_subdomain:bool=False, period:Optional[MonitoringTargetPeriod]=ScraperAPI.MONITORING_PERIOD_LAST_24H, start:Optional[datetime.datetime]=None, end:Optional[datetime.datetime]=None, include_webhook:bool=False, ): return self._monitoring_request( '/crawl/monitoring/metrics/target', self._build_target_params(domain, group_subdomain, period, start, end, include_webhook), ) def get_extraction_monitoring_metrics(self,
format: str = 'structured',
period: str | None = None,
aggregation: List[Literal['account', 'project', 'target']] | None = None,
include_webhook: bool = False)-
Expand source code
def get_extraction_monitoring_metrics( self, format:str=ScraperAPI.MONITORING_DATA_FORMAT_STRUCTURED, period:Optional[str]=None, aggregation:Optional[List[MonitoringAggregation]]=None, include_webhook:bool=False, ): return self._monitoring_request( '/extraction/monitoring/metrics', self._build_metrics_params(format, period, aggregation, include_webhook), ) def get_extraction_monitoring_target_metrics(self,
domain: str,
group_subdomain: bool = False,
period: Literal['subscription', 'last7d', 'last24h', 'last1h', 'last5m'] | None = 'last24h',
start: datetime.datetime | None = None,
end: datetime.datetime | None = None,
include_webhook: bool = False)-
Expand source code
def get_extraction_monitoring_target_metrics( self, domain:str, group_subdomain:bool=False, period:Optional[MonitoringTargetPeriod]=ScraperAPI.MONITORING_PERIOD_LAST_24H, start:Optional[datetime.datetime]=None, end:Optional[datetime.datetime]=None, include_webhook:bool=False, ): return self._monitoring_request( '/extraction/monitoring/metrics/target', self._build_target_params(domain, group_subdomain, period, start, end, include_webhook), ) def get_monitoring_metrics(self,
format: str = 'structured',
period: str | None = None,
aggregation: List[Literal['account', 'project', 'target']] | None = None,
include_webhook: bool = False)-
Expand source code
def get_monitoring_metrics( self, format:str=ScraperAPI.MONITORING_DATA_FORMAT_STRUCTURED, period:Optional[str]=None, aggregation:Optional[List[MonitoringAggregation]]=None, include_webhook:bool=False, ): return self._monitoring_request( '/scrape/monitoring/metrics', self._build_metrics_params(format, period, aggregation, include_webhook), ) def get_monitoring_target_metrics(self,
domain: str,
group_subdomain: bool = False,
period: Literal['subscription', 'last7d', 'last24h', 'last1h', 'last5m'] | None = 'last24h',
start: datetime.datetime | None = None,
end: datetime.datetime | None = None,
include_webhook: bool = False)-
Expand source code
def get_monitoring_target_metrics( self, domain:str, group_subdomain:bool=False, period:Optional[MonitoringTargetPeriod]=ScraperAPI.MONITORING_PERIOD_LAST_24H, start:Optional[datetime.datetime]=None, end:Optional[datetime.datetime]=None, include_webhook:bool=False, ): return self._monitoring_request( '/scrape/monitoring/metrics/target', self._build_target_params(domain, group_subdomain, period, start, end, include_webhook), ) def get_screenshot_monitoring_metrics(self,
format: str = 'structured',
period: str | None = None,
aggregation: List[Literal['account', 'project', 'target']] | None = None,
include_webhook: bool = False)-
Expand source code
def get_screenshot_monitoring_metrics( self, format:str=ScraperAPI.MONITORING_DATA_FORMAT_STRUCTURED, period:Optional[str]=None, aggregation:Optional[List[MonitoringAggregation]]=None, include_webhook:bool=False, ): return self._monitoring_request( '/screenshot/monitoring/metrics', self._build_metrics_params(format, period, aggregation, include_webhook), ) def get_screenshot_monitoring_target_metrics(self,
domain: str,
group_subdomain: bool = False,
period: Literal['subscription', 'last7d', 'last24h', 'last1h', 'last5m'] | None = 'last24h',
start: datetime.datetime | None = None,
end: datetime.datetime | None = None,
include_webhook: bool = False)-
Expand source code
def get_screenshot_monitoring_target_metrics( self, domain:str, group_subdomain:bool=False, period:Optional[MonitoringTargetPeriod]=ScraperAPI.MONITORING_PERIOD_LAST_24H, start:Optional[datetime.datetime]=None, end:Optional[datetime.datetime]=None, include_webhook:bool=False, ): return self._monitoring_request( '/screenshot/monitoring/metrics/target', self._build_target_params(domain, group_subdomain, period, start, end, include_webhook), ) def open(self)-
Expand source code
def open(self): if self.http_session is None: self.http_session = Session() self.http_session.verify = self.verify self.http_session.timeout = (self.connect_timeout, self.default_read_timeout) self.http_session.params['key'] = self.key self.http_session.headers['accept-encoding'] = self.body_handler.content_encoding self.http_session.headers['accept'] = self.body_handler.accept self.http_session.headers['user-agent'] = self.ua def resilient_scrape(self,
scrape_config: ScrapeConfig,
retry_on_errors: Set[Exception] | None = None,
retry_on_status_code: List[int] | None = None,
tries: int = 5,
delay: int = 20) ‑> ScrapeApiResponse-
Expand source code
def resilient_scrape( self, scrape_config:ScrapeConfig, retry_on_errors:Optional[Set[Exception]]=None, retry_on_status_code:Optional[List[int]]=None, tries: int = 5, delay: int = 20, ) -> ScrapeApiResponse: if retry_on_errors is None: retry_on_errors = {ScrapflyError} assert isinstance(retry_on_errors, set), 'retry_on_errors is not a set()' @backoff.on_exception(backoff.expo, exception=tuple(retry_on_errors), max_tries=tries, max_time=delay) def inner() -> ScrapeApiResponse: try: return self.scrape(scrape_config=scrape_config) except (UpstreamHttpClientError, UpstreamHttpServerError) as e: if retry_on_status_code is not None and e.api_response: if e.api_response.upstream_status_code in retry_on_status_code: raise e else: return e.api_response raise e return inner() def save_scrape_screenshot(self,
api_response: ScrapeApiResponse,
name: str,
path: str | None = None)-
Expand source code
def save_scrape_screenshot(self, api_response:ScrapeApiResponse, name:str, path:Optional[str]=None): """ Save a screenshot from a scrape result :param api_response: ScrapeApiResponse :param name: str - name of the screenshot given in the scrape config :param path: Optional[str] """ if not api_response.scrape_result['screenshots']: raise RuntimeError('Screenshot %s do no exists' % name) try: api_response.scrape_result['screenshots'][name] except KeyError: raise RuntimeError('Screenshot %s do no exists' % name) screenshot_response = self._http_handler( method='GET', url=api_response.scrape_result['screenshots'][name]['url'], params={'key': self.key}, verify=self.verify ) screenshot_response.raise_for_status() if not name.endswith('.jpg'): name += '.jpg' api_response.sink(path=path, name=name, content=screenshot_response.content)Save a screenshot from a scrape result :param api_response: ScrapeApiResponse :param name: str - name of the screenshot given in the scrape config :param path: Optional[str]
def save_screenshot(self,
screenshot_api_response: ScreenshotApiResponse,
name: str,
path: str | None = None)-
Expand source code
def save_screenshot(self, screenshot_api_response:ScreenshotApiResponse, name:str, path:Optional[str]=None): """ Save a screenshot from a screenshot API response :param api_response: ScreenshotApiResponse :param name: str - name of the screenshot to save as :param path: Optional[str] """ if screenshot_api_response.screenshot_success is not True: raise RuntimeError('Screenshot was not successful') if not screenshot_api_response.image: raise RuntimeError('Screenshot binary does not exist') content = screenshot_api_response.image extension_name = screenshot_api_response.metadata['extension_name'] if path: os.makedirs(path, exist_ok=True) file_path = os.path.join(path, f'{name}.{extension_name}') else: file_path = f'{name}.{extension_name}' if isinstance(content, bytes): content = BytesIO(content) with open(file_path, 'wb') as f: shutil.copyfileobj(content, f, length=131072)Save a screenshot from a screenshot API response :param api_response: ScreenshotApiResponse :param name: str - name of the screenshot to save as :param path: Optional[str]
def scrape(self,
scrape_config: ScrapeConfig,
no_raise: bool = False) ‑> ScrapeApiResponse-
Expand source code
@backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def scrape(self, scrape_config:ScrapeConfig, no_raise:bool=False) -> ScrapeApiResponse: """ Scrape a website :param scrape_config: ScrapeConfig :param no_raise: bool - if True, do not raise exception on error while the api response is a ScrapflyError for seamless integration :return: ScrapeApiResponse If you use no_raise=True, make sure to check the api_response.scrape_result.error attribute to handle the error. If the error is not none, you will get the following structure for example 'error': { 'code': 'ERR::ASP::SHIELD_PROTECTION_FAILED', 'message': 'The ASP shield failed to solve the challenge against the anti scrapping protection - heuristic_engine bypass failed, please retry in few seconds', 'retryable': False, 'http_code': 422, 'links': { 'Checkout ASP documentation': 'https://scrapfly.io/docs/scrape-api/anti-scraping-protection#maximize_success_rate', 'Related Error Doc': 'https://scrapfly.io/docs/scrape-api/error/ERR::ASP::SHIELD_PROTECTION_FAILED' } } """ try: logger.debug('--> %s Scrapping %s' % (scrape_config.method, scrape_config.url)) request_data = self._scrape_request(scrape_config=scrape_config) response = self._http_handler(**request_data) if scrape_config.proxified_response is True: # Proxified mode: the API returns the raw upstream response # (target's status, headers, body) instead of the JSON # envelope. Error restoration: if X-Scrapfly-Reject-Code is # present, the scrape failed and the SDK must raise a typed # error with the code/message/retryable from the headers. reject_code = response.headers.get('X-Scrapfly-Reject-Code') if reject_code: from scrapfly.errors import HttpError reject_desc = response.headers.get('X-Scrapfly-Reject-Description', '') reject_retryable = response.headers.get('X-Scrapfly-Reject-Retryable', 'false').lower() == 'true' retry_after = None if reject_retryable: try: retry_after = int(response.headers.get('Retry-After', '0')) except (ValueError, TypeError): retry_after = None raise HttpError( request=response.request, response=response, code=reject_code, http_status_code=response.status_code, message=reject_desc, is_retryable=reject_retryable, retry_delay=retry_after, ) self.reporter.report(scrape_api_response=None) return response scrape_api_response = self._handle_response(response=response, scrape_config=scrape_config) self.reporter.report(scrape_api_response=scrape_api_response) return scrape_api_response except BaseException as e: self.reporter.report(error=e) if no_raise and isinstance(e, ScrapflyError) and e.api_response is not None: return e.api_response raise eScrape a website :param scrape_config: ScrapeConfig :param no_raise: bool - if True, do not raise exception on error while the api response is a ScrapflyError for seamless integration :return: ScrapeApiResponse
If you use no_raise=True, make sure to check the api_response.scrape_result.error attribute to handle the error. If the error is not none, you will get the following structure for example
'error': { 'code': 'ERR::ASP::SHIELD_PROTECTION_FAILED', 'message': 'The ASP shield failed to solve the challenge against the anti scrapping protection - heuristic_engine bypass failed, please retry in few seconds', 'retryable': False, 'http_code': 422, 'links': { 'Checkout ASP documentation': 'https://scrapfly.io/docs/scrape-api/anti-scraping-protection#maximize_success_rate', 'Related Error Doc': 'https://scrapfly.io/docs/scrape-api/error/ERR::ASP::SHIELD_PROTECTION_FAILED' } }
def scrape_batch(self,
scrape_configs: List[ScrapeConfig],
format: Literal['json', 'msgpack'] | None = None) ‑> Iterator[Tuple[str, ScrapeApiResponse | ScrapflyError]]-
Expand source code
@backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def scrape_batch( self, scrape_configs: List[ScrapeConfig], format: Optional[Literal['json', 'msgpack']] = None, ) -> Iterator[Tuple[str, Union[ScrapeApiResponse, ScrapflyError]]]: """ Scrape up to 100 URLs in one batch request and stream results back as each scrape completes. Iterator yields ``(correlation_id, result)`` tuples where ``result`` is either a :class:`ScrapeApiResponse` on success or a :class:`ScrapflyError` on per-scrape failure. Results arrive **out of order** — whichever scrape finishes first is yielded first. Use ``correlation_id`` (set on every ``ScrapeConfig``) to match parts back to the originating config on the client side. Every config MUST carry a unique ``correlation_id``; a missing/duplicate value is detected client-side before the batch is sent. :param format: wire format for per-part response bodies. Defaults to the SDK's negotiated format (``msgpack`` when the ``msgpack`` package is installed, ``json`` otherwise). Pass ``'json'`` or ``'msgpack'`` to override. """ from .batch import ( iter_batch_parts, decode_part_body, is_api_error_part, error_from_api_error_part, _build_proxified_response_from_part, ) if not scrape_configs: raise ScrapflyError( "scrape_batch: configs list is empty", code="ERR::SCRAPE::BATCH_CONFIG", http_status_code=400, ) if len(scrape_configs) > 100: raise ScrapflyError( f"scrape_batch: max 100 configs per batch (got {len(scrape_configs)})", code="ERR::SCRAPE::BATCH_CONFIG", http_status_code=400, ) seen_correlations: Dict[str, int] = {} body_configs: List[Dict[str, Any]] = [] config_by_correlation: Dict[str, ScrapeConfig] = {} for idx, cfg in enumerate(scrape_configs): if not getattr(cfg, "correlation_id", None): raise ScrapflyError( f"scrape_batch: configs[{idx}] is missing correlation_id " "(required for matching streamed parts)", code="ERR::SCRAPE::BATCH_CONFIG", http_status_code=422, ) if cfg.correlation_id in seen_correlations: raise ScrapflyError( f"scrape_batch: correlation_id {cfg.correlation_id!r} reused by " f"configs[{seen_correlations[cfg.correlation_id]}] and configs[{idx}]", code="ERR::SCRAPE::BATCH_CONFIG", http_status_code=422, ) seen_correlations[cfg.correlation_id] = idx config_by_correlation[cfg.correlation_id] = cfg # Drop `key` (batch key goes in the URL); pass everything # else as a flat query-param dict. The server feeds each # entry through NewScrapeConfigFromRequest identically to # a /scrape call, so the wire contract is identical. params = cfg.to_api_params(key=self.key) params.pop("key", None) body_configs.append(params) import json as _json payload = _json.dumps({"configs": body_configs}).encode("utf-8") if format == 'msgpack': accept_header = 'application/msgpack' elif format == 'json': accept_header = 'application/json' else: accept_header = self.body_handler.accept request = { "method": "POST", "url": self.host + "/scrape/batch", "params": {"key": self.key}, "data": payload, "headers": { "content-type": "application/json", "accept-encoding": self.body_handler.content_encoding, "accept": accept_header, "user-agent": self.ua, }, "timeout": (self.connect_timeout, self.web_scraping_api_read_timeout), "verify": self.verify, "stream": True, } # Own the session for the life of the streaming batch so its # connection pool closes whether the generator is fully consumed, # errors mid-stream, or is abandoned (finally runs on GC/close()). batch_session = requests.Session() batch_session.verify = self.verify try: response = batch_session.request( method=request["method"], url=request["url"], params=request["params"], data=request["data"], headers=request["headers"], timeout=request["timeout"], stream=request["stream"], ) if response.status_code != 200: # Batch-level error (plan gate, validation, insufficient # concurrency, etc.). Response is a single JSON body, not # multipart. try: body = response.json() except Exception: body = {"message": response.text, "code": "ERR::API::INTERNAL_ERROR"} err_code = body.get("code", "ERR::API::INTERNAL_ERROR") err_msg = body.get("message", "") or body.get("reason", "") retry_after = None try: retry_after = int(response.headers.get("Retry-After", "0")) or None except (TypeError, ValueError): pass raise HttpError( request=response.request, response=response, code=err_code, http_status_code=response.status_code, message=err_msg, is_retryable=body.get("retryable", False), retry_delay=retry_after, ) for part_headers, part_body in iter_batch_parts(response): correlation_id = part_headers.get("x-scrapfly-correlation-id", "") cfg = config_by_correlation.get(correlation_id, scrape_configs[0]) # Proxified-response parts: the part body is the raw # upstream bytes, not a JSON envelope. Surface a native # requests.Response synthesized from the part headers + # body so callers get the same shape as a single # proxified scrape. if part_headers.get("x-scrapfly-proxified") == "true": try: prox_response = _build_proxified_response_from_part( part_headers, part_body, originating_request=response.request, ) except Exception as prox_err: yield correlation_id, ScrapflyError( f"scrape_batch: failed to build proxified response for correlation_id={correlation_id!r}: {prox_err}", code="ERR::API::INTERNAL_ERROR", http_status_code=500, ) continue yield correlation_id, prox_response continue # EncoderError subclasses BaseException — catch it explicitly. try: parsed = decode_part_body(part_headers, part_body, self.body_handler) except (EncoderError, Exception) as decode_err: yield correlation_id, ScrapflyError( f"scrape_batch: failed to decode part for correlation_id={correlation_id!r}: {decode_err}", code="ERR::API::INTERNAL_ERROR", http_status_code=500, ) continue # API-generated error parts carry an error body instead of # the scrape envelope — surface them as typed per-part errors. if is_api_error_part(parsed, part_headers): try: part_error = error_from_api_error_part(parsed, part_headers, response.request) except Exception as factory_err: part_error = ScrapflyError( f"scrape_batch: malformed error part for correlation_id={correlation_id!r}: {factory_err}", code="ERR::API::INTERNAL_ERROR", http_status_code=500, ) yield correlation_id, part_error continue part_result = None try: api_response = ScrapeApiResponse( response=response, request=response.request, api_result=parsed, scrape_config=cfg, large_object_handler=self._handle_scrape_large_objects, ) # Don't auto-raise on upstream error — per-part errors # are surfaced via the yielded tuple, not exceptions. api_response.raise_for_result(raise_on_upstream_error=False) part_result = api_response except ScrapflyError as scrape_err: part_result = scrape_err except (EncoderError, Exception) as part_err: part_result = ScrapflyError( f"scrape_batch: failed to process part for correlation_id={correlation_id!r}: {part_err}", code="ERR::API::INTERNAL_ERROR", http_status_code=500, ) yield correlation_id, part_result finally: batch_session.close()Scrape up to 100 URLs in one batch request and stream results back as each scrape completes. Iterator yields
(correlation_id, result)tuples whereresultis either a :class:ScrapeApiResponseon success or a :class:ScrapflyErroron per-scrape failure.Results arrive out of order — whichever scrape finishes first is yielded first. Use
correlation_id(set on everyScrapeConfig) to match parts back to the originating config on the client side.Every config MUST carry a unique
correlation_id; a missing/duplicate value is detected client-side before the batch is sent.:param format: wire format for per-part response bodies. Defaults to the SDK's negotiated format (
msgpackwhen themsgpackpackage is installed,jsonotherwise). Pass'json'or'msgpack'to override. def screenshot(self,
screenshot_config: ScreenshotConfig,
no_raise: bool = False) ‑> ScreenshotApiResponse-
Expand source code
@backoff.on_exception(backoff.expo, exception=NetworkError, max_tries=5) def screenshot(self, screenshot_config:ScreenshotConfig, no_raise:bool=False) -> ScreenshotApiResponse: """ Take a screenshot :param screenshot_config: ScrapeConfig :param no_raise: bool - if True, do not raise exception on error while the screenshot api response is a ScrapflyError for seamless integration :return: str If you use no_raise=True, make sure to check the screenshot_api_response.error attribute to handle the error. If the error is not none, you will get the following structure for example 'error': { 'code': 'ERR::SCREENSHOT::UNABLE_TO_TAKE_SCREENSHOT', 'message': 'For some reason we were unable to take the screenshot', 'http_code': 422, 'links': { 'Checkout the related doc: https://scrapfly.io/docs/screenshot-api/error/ERR::SCREENSHOT::UNABLE_TO_TAKE_SCREENSHOT' } } """ try: logger.debug('--> %s Screenshoting' % (screenshot_config.url)) request_data = self._screenshot_request(screenshot_config=screenshot_config) response = self._http_handler(**request_data) screenshot_api_response = self._handle_screenshot_response(response=response, screenshot_config=screenshot_config) return screenshot_api_response except BaseException as e: self.reporter.report(error=e) if no_raise and isinstance(e, ScrapflyError) and e.api_response is not None: return e.api_response raise eTake a screenshot :param screenshot_config: ScrapeConfig :param no_raise: bool - if True, do not raise exception on error while the screenshot api response is a ScrapflyError for seamless integration :return: str
If you use no_raise=True, make sure to check the screenshot_api_response.error attribute to handle the error. If the error is not none, you will get the following structure for example
'error': { 'code': 'ERR::SCREENSHOT::UNABLE_TO_TAKE_SCREENSHOT', 'message': 'For some reason we were unable to take the screenshot', 'http_code': 422, 'links': { 'Checkout the related doc: https://scrapfly.io/docs/screenshot-api/error/ERR::SCREENSHOT::UNABLE_TO_TAKE_SCREENSHOT' } }
def sink(self,
api_response: ScrapeApiResponse,
content: str | bytes | None = None,
path: str | None = None,
name: str | None = None,
file:| _io.BytesIO | None = None) ‑> str -
Expand source code
def sink(self, api_response:ScrapeApiResponse, content:Optional[Union[str, bytes]]=None, path: Optional[str] = None, name: Optional[str] = None, file: Optional[Union[TextIO, BytesIO]] = None) -> str: scrape_result = api_response.result['result'] scrape_config = api_response.result['config'] file_content = content or scrape_result['content'] file_path = None file_extension = None if name: name_parts = name.split('.') if len(name_parts) > 1: file_extension = name_parts[-1] if not file: if file_extension is None: try: mime_type = scrape_result['response_headers']['content-type'] except KeyError: mime_type = 'application/octet-stream' if ';' in mime_type: mime_type = mime_type.split(';')[0] file_extension = '.' + mime_type.split('/')[1] if not name: name = scrape_config['url'].split('/')[-1] if name.find(file_extension) == -1: name += file_extension file_path = path + '/' + name if path else name if file_path == file_extension: url = re.sub(r'(https|http)?://', '', api_response.config['url']).replace('/', '-') if url[-1] == '-': url = url[:-1] url += file_extension file_path = url file = open(file_path, 'wb') if isinstance(file_content, str): file_content = BytesIO(file_content.encode('utf-8')) elif isinstance(file_content, bytes): file_content = BytesIO(file_content) file_content.seek(0) with file as f: shutil.copyfileobj(file_content, f, length=131072) logger.info('file %s created' % file_path) return file_path def start_crawl(self,
crawler_config: CrawlerConfig) ‑> CrawlerStartResponse-
Expand source code
@backoff.on_exception(backoff.expo, exception=ConnectionError, max_tries=5) def start_crawl(self, crawler_config: CrawlerConfig) -> CrawlerStartResponse: """ Start a crawler job :param crawler_config: CrawlerConfig :return: CrawlerStartResponse with UUID and initial status Example: ```python from scrapfly import ScrapflyClient, CrawlerConfig client = ScrapflyClient(key='YOUR_API_KEY') config = CrawlerConfig( url='https://example.com', page_limit=100, max_depth=3 ) response = client.start_crawl(config) print(f"Crawler started: {response.uuid}") ``` """ # POST /crawl accepts two body formats: # - application/json: the entire crawler configuration as JSON. # Used for seed-URL crawls and remote_url_list crawls. # - multipart/form-data: a 'config' JSON part and a 'urls' text part # (one URL per line). Used only when the caller provides an # in-memory url_list, so we can stream it as a file payload # instead of inlining it into the JSON body. parts = crawler_config.to_multipart_parts() urls_blob = parts['urls'] query_params = {'key': self.key} timeout = (self.connect_timeout, self.DEFAULT_CRAWLER_API_READ_TIMEOUT) url = f'{self.host}/crawl' logger.debug(f"Crawler API POST {url}?key=***") if urls_blob is not None: config_body = json.dumps(parts['config']).encode('utf-8') files = { 'config': ('config.json', config_body, 'application/json'), 'urls': ('urls.txt', urls_blob.encode('utf-8'), 'text/plain'), } logger.debug( f"Crawler API multipart config: {parts['config']} ; " f"urls part: {len(urls_blob.splitlines())} URL(s)" ) response = self._http_handler( method='POST', url=url, params=query_params, files=files, timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) else: logger.debug(f"Crawler API body: {parts['config']}") response = self._http_handler( method='POST', url=url, params=query_params, json=parts['config'], timeout=timeout, headers={'User-Agent': self.ua}, verify=self.verify ) if response.status_code not in (200, 201): # Log error details for debugging try: error_detail = response.json() except (ValueError, Exception): error_detail = response.text logger.debug(f"Crawler API error ({response.status_code}): {error_detail}") self._handle_crawler_error_response(response) result = response.json() return CrawlerStartResponse(result)Start a crawler job
:param crawler_config: CrawlerConfig :return: CrawlerStartResponse with UUID and initial status
Example
from scrapfly import ScrapflyClient, CrawlerConfig client = ScrapflyClient(key='YOUR_API_KEY') config = CrawlerConfig( url='https://example.com', page_limit=100, max_depth=3 ) response = client.start_crawl(config) print(f"Crawler started: {response.uuid}")
Inherited members
class ScrapflyCrawlerError (message: str,
code: str,
http_status_code: int,
resource: str | None = None,
is_retryable: bool = False,
retry_delay: int | None = None,
retry_times: int | None = None,
documentation_url: str | None = None,
api_response: ForwardRef('ApiResponse') | None = None)-
Expand source code
class ScrapflyCrawlerError(CrawlerError): """Exception raised when a crawler job fails or is cancelled""" passException raised when a crawler job fails or is cancelled
Ancestors
- CrawlerError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class ScrapflyError (message: str,
code: str,
http_status_code: int,
resource: str | None = None,
is_retryable: bool = False,
retry_delay: int | None = None,
retry_times: int | None = None,
documentation_url: str | None = None,
api_response: ForwardRef('ApiResponse') | None = None)-
Expand source code
class ScrapflyError(Exception): KIND_HTTP_BAD_RESPONSE = 'HTTP_BAD_RESPONSE' KIND_SCRAPFLY_ERROR = 'SCRAPFLY_ERROR' RESOURCE_PROXY = 'PROXY' RESOURCE_THROTTLE = 'THROTTLE' RESOURCE_SCRAPE = 'SCRAPE' RESOURCE_ASP = 'ASP' RESOURCE_SCHEDULE = 'SCHEDULE' RESOURCE_WEBHOOK = 'WEBHOOK' RESOURCE_SESSION = 'SESSION' def __init__( self, message: str, code: str, http_status_code: int, resource: Optional[str]=None, is_retryable: bool = False, retry_delay: Optional[int] = None, retry_times: Optional[int] = None, documentation_url: Optional[str] = None, api_response: Optional['ApiResponse'] = None ): self.message = message self.code = code self.retry_delay = retry_delay self.retry_times = retry_times self.resource = resource self.is_retryable = is_retryable self.documentation_url = documentation_url self.api_response = api_response self.http_status_code = http_status_code super().__init__(self.message, str(self.code)) def __str__(self): message = self.message if self.documentation_url is not None: message += '. Learn more: %s' % self.documentation_url return messageCommon base class for all non-exit exceptions.
Ancestors
- builtins.Exception
- builtins.BaseException
Subclasses
- CrawlerError
- scrapfly.errors.ExtraUsageForbidden
- scrapfly.errors.HttpError
Class variables
var KIND_HTTP_BAD_RESPONSEvar KIND_SCRAPFLY_ERRORvar RESOURCE_ASPvar RESOURCE_PROXYvar RESOURCE_SCHEDULEvar RESOURCE_SCRAPEvar RESOURCE_SESSIONvar RESOURCE_THROTTLEvar RESOURCE_WEBHOOK
class ScrapflyProxyError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class ScrapflyProxyError(ScraperAPIError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.ScraperAPIError
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class ScrapflyScheduleError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class ScrapflyScheduleError(ScraperAPIError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.ScraperAPIError
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class ScrapflyScrapeError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class ScrapflyScrapeError(ScraperAPIError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.ScraperAPIError
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class ScrapflySessionError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class ScrapflySessionError(ScraperAPIError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.ScraperAPIError
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class ScrapflyThrottleError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class ScrapflyThrottleError(ScraperAPIError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.ScraperAPIError
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class ScrapflyWebhookError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class ScrapflyWebhookError(ScraperAPIError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.ScraperAPIError
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class ScreenshotAPIError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class ScreenshotAPIError(HttpError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class ScreenshotApiResponse (request: requests.models.Request,
response: requests.models.Response,
screenshot_config: ScreenshotConfig,
api_result: bytes | None = None)-
Expand source code
class ScreenshotApiResponse(ApiResponse): def __init__(self, request: Request, response: Response, screenshot_config: ScreenshotConfig, api_result: Optional[bytes] = None): super().__init__(request, response) self.screenshot_config = screenshot_config self.result = self.handle_api_result(api_result) @property def image(self) -> Optional[str]: binary = self.result.get('result', None) if binary is None: return '' return binary @property def metadata(self) -> Optional[Dict]: if not self.image: return {} content_type = self.response.headers.get('content-type') extension_name = content_type[content_type.find('/') + 1:].split(';')[0] return { 'extension_name': extension_name, 'upstream-status-code': self.response.headers.get('X-Scrapfly-Upstream-Http-Code'), 'upstream-url': self.response.headers.get('X-Scrapfly-Upstream-Url') } @property def screenshot_success(self) -> bool: if not self.image: return False return True @property def error(self) -> Optional[Dict]: if self.image: return None if self.screenshot_success is False: return self.result def _is_api_error(self, api_result: Dict) -> bool: if api_result is None: return True return 'error_id' in api_result def handle_api_result(self, api_result: bytes) -> FrozenDict: if self._is_api_error(api_result=api_result) is True: return FrozenDict(api_result) return api_result def raise_for_result(self, raise_on_upstream_error=True, error_class=ScreenshotAPIError): super().raise_for_result(raise_on_upstream_error=raise_on_upstream_error, error_class=error_class)Ancestors
Instance variables
prop error : Dict | None-
Expand source code
@property def error(self) -> Optional[Dict]: if self.image: return None if self.screenshot_success is False: return self.result prop image : str | None-
Expand source code
@property def image(self) -> Optional[str]: binary = self.result.get('result', None) if binary is None: return '' return binary prop metadata : Dict | None-
Expand source code
@property def metadata(self) -> Optional[Dict]: if not self.image: return {} content_type = self.response.headers.get('content-type') extension_name = content_type[content_type.find('/') + 1:].split(';')[0] return { 'extension_name': extension_name, 'upstream-status-code': self.response.headers.get('X-Scrapfly-Upstream-Http-Code'), 'upstream-url': self.response.headers.get('X-Scrapfly-Upstream-Url') } prop screenshot_success : bool-
Expand source code
@property def screenshot_success(self) -> bool: if not self.image: return False return True
Methods
def handle_api_result(self, api_result: bytes) ‑> FrozenDict-
Expand source code
def handle_api_result(self, api_result: bytes) -> FrozenDict: if self._is_api_error(api_result=api_result) is True: return FrozenDict(api_result) return api_result def raise_for_result(self,
raise_on_upstream_error=True,
error_class=scrapfly.errors.ScreenshotAPIError)-
Expand source code
def raise_for_result(self, raise_on_upstream_error=True, error_class=ScreenshotAPIError): super().raise_for_result(raise_on_upstream_error=raise_on_upstream_error, error_class=error_class)
Inherited members
class ScreenshotConfig (url: str,
format: Format | None = None,
capture: str | None = None,
resolution: str | None = None,
country: str | None = None,
timeout: int | None = None,
rendering_wait: int | None = None,
wait_for_selector: str | None = None,
options: List[Options] | None = None,
auto_scroll: bool | None = None,
js: str | None = None,
cache: bool | None = None,
cache_ttl: int | None = None,
cache_clear: bool | None = None,
vision_deficiency: VisionDeficiency | None = None,
webhook: str | None = None,
raise_on_upstream_error: bool = True)-
Expand source code
class ScreenshotConfig(BaseApiConfig): url: str format: Optional[Format] = None capture: Optional[str] = None resolution: Optional[str] = None country: Optional[str] = None timeout: Optional[int] = None # in milliseconds rendering_wait: Optional[int] = None # in milliseconds wait_for_selector: Optional[str] = None options: Optional[List[Options]] = None auto_scroll: Optional[bool] = None js: Optional[str] = None cache: Optional[bool] = None cache_ttl: Optional[int] = None cache_clear: Optional[bool] = None webhook: Optional[str] = None raise_on_upstream_error: bool = True def __init__( self, url: str, format: Optional[Format] = None, capture: Optional[str] = None, resolution: Optional[str] = None, country: Optional[str] = None, timeout: Optional[int] = None, # in milliseconds rendering_wait: Optional[int] = None, # in milliseconds wait_for_selector: Optional[str] = None, options: Optional[List[Options]] = None, auto_scroll: Optional[bool] = None, js: Optional[str] = None, cache: Optional[bool] = None, cache_ttl: Optional[int] = None, cache_clear: Optional[bool] = None, vision_deficiency: Optional[VisionDeficiency] = None, webhook: Optional[str] = None, raise_on_upstream_error: bool = True ): assert(type(url) is str) self.url = url self.key = None self.format = format self.capture = capture self.resolution = resolution self.country = country self.timeout = timeout self.rendering_wait = rendering_wait self.wait_for_selector = wait_for_selector self.options = [Options(flag) for flag in options] if options else None self.auto_scroll = auto_scroll self.js = js self.cache = cache self.cache_ttl = cache_ttl self.cache_clear = cache_clear self.vision_deficiency = vision_deficiency self.webhook = webhook self.raise_on_upstream_error = raise_on_upstream_error def to_api_params(self, key:str) -> Dict: params = { 'key': self.key or key, 'url': self.url } if self.format: params['format'] = Format(self.format).value if self.capture: params['capture'] = self.capture if self.resolution: params['resolution'] = self.resolution if self.country is not None: params['country'] = self.country if self.timeout is not None: params['timeout'] = self.timeout if self.rendering_wait is not None: params['rendering_wait'] = self.rendering_wait if self.wait_for_selector is not None: params['wait_for_selector'] = self.wait_for_selector if self.options is not None: params["options"] = ",".join(flag.value for flag in self.options) if self.auto_scroll is not None: params['auto_scroll'] = self._bool_to_http(self.auto_scroll) if self.js: params['js'] = base64.urlsafe_b64encode(self.js.encode('utf-8')).decode('utf-8') if self.cache is not None: params['cache'] = self._bool_to_http(self.cache) if self.cache_ttl is not None: params['cache_ttl'] = self.cache_ttl if self.cache_clear is not None: params['cache_clear'] = self._bool_to_http(self.cache_clear) else: if self.cache_ttl is not None: logging.warning('Params "cache_ttl" is ignored. Works only if cache is enabled') if self.cache_clear is not None: logging.warning('Params "cache_clear" is ignored. Works only if cache is enabled') if self.vision_deficiency is not None: params['vision_deficiency'] = self.vision_deficiency.value if self.webhook is not None: params['webhook_name'] = self.webhook return params def to_dict(self) -> Dict: """ Export the ScreenshotConfig instance to a plain dictionary. """ return { 'url': self.url, 'format': Format(self.format).value if self.format else None, 'capture': self.capture, 'resolution': self.resolution, 'country': self.country, 'timeout': self.timeout, 'rendering_wait': self.rendering_wait, 'wait_for_selector': self.wait_for_selector, 'options': [Options(option).value for option in self.options] if self.options else None, 'auto_scroll': self.auto_scroll, 'js': self.js, 'cache': self.cache, 'cache_ttl': self.cache_ttl, 'cache_clear': self.cache_clear, 'vision_deficiency': self.vision_deficiency.value if self.vision_deficiency else None, 'webhook': self.webhook, 'raise_on_upstream_error': self.raise_on_upstream_error } @staticmethod def from_dict(screenshot_config_dict: Dict) -> 'ScreenshotConfig': """Create a ScreenshotConfig instance from a dictionary.""" url = screenshot_config_dict.get('url', None) format = screenshot_config_dict.get('format', None) format = Format(format) if format else None capture = screenshot_config_dict.get('capture', None) resolution = screenshot_config_dict.get('resolution', None) country = screenshot_config_dict.get('country', None) timeout = screenshot_config_dict.get('timeout', None) rendering_wait = screenshot_config_dict.get('rendering_wait', None) wait_for_selector = screenshot_config_dict.get('wait_for_selector', None) options = screenshot_config_dict.get('options', None) options = [Options(option) for option in options] if options else None auto_scroll = screenshot_config_dict.get('auto_scroll', None) js = screenshot_config_dict.get('js', None) cache = screenshot_config_dict.get('cache', None) cache_ttl = screenshot_config_dict.get('cache_ttl', None) cache_clear = screenshot_config_dict.get('cache_clear', None) vision_deficiency = screenshot_config_dict.get('vision_deficiency', None) webhook = screenshot_config_dict.get('webhook', None) raise_on_upstream_error = screenshot_config_dict.get('raise_on_upstream_error', True) return ScreenshotConfig( url=url, format=format, capture=capture, resolution=resolution, country=country, timeout=timeout, rendering_wait=rendering_wait, wait_for_selector=wait_for_selector, options=options, auto_scroll=auto_scroll, js=js, cache=cache, cache_ttl=cache_ttl, cache_clear=cache_clear, vision_deficiency=vision_deficiency, webhook=webhook, raise_on_upstream_error=raise_on_upstream_error )Ancestors
Class variables
var auto_scroll : bool | Nonevar cache : bool | Nonevar cache_clear : bool | Nonevar cache_ttl : int | Nonevar capture : str | Nonevar country : str | Nonevar format : Format | Nonevar js : str | Nonevar options : List[Options] | Nonevar raise_on_upstream_error : boolvar rendering_wait : int | Nonevar resolution : str | Nonevar timeout : int | Nonevar url : strvar wait_for_selector : str | Nonevar webhook : str | None
Static methods
def from_dict(screenshot_config_dict: Dict) ‑> ScreenshotConfig-
Expand source code
@staticmethod def from_dict(screenshot_config_dict: Dict) -> 'ScreenshotConfig': """Create a ScreenshotConfig instance from a dictionary.""" url = screenshot_config_dict.get('url', None) format = screenshot_config_dict.get('format', None) format = Format(format) if format else None capture = screenshot_config_dict.get('capture', None) resolution = screenshot_config_dict.get('resolution', None) country = screenshot_config_dict.get('country', None) timeout = screenshot_config_dict.get('timeout', None) rendering_wait = screenshot_config_dict.get('rendering_wait', None) wait_for_selector = screenshot_config_dict.get('wait_for_selector', None) options = screenshot_config_dict.get('options', None) options = [Options(option) for option in options] if options else None auto_scroll = screenshot_config_dict.get('auto_scroll', None) js = screenshot_config_dict.get('js', None) cache = screenshot_config_dict.get('cache', None) cache_ttl = screenshot_config_dict.get('cache_ttl', None) cache_clear = screenshot_config_dict.get('cache_clear', None) vision_deficiency = screenshot_config_dict.get('vision_deficiency', None) webhook = screenshot_config_dict.get('webhook', None) raise_on_upstream_error = screenshot_config_dict.get('raise_on_upstream_error', True) return ScreenshotConfig( url=url, format=format, capture=capture, resolution=resolution, country=country, timeout=timeout, rendering_wait=rendering_wait, wait_for_selector=wait_for_selector, options=options, auto_scroll=auto_scroll, js=js, cache=cache, cache_ttl=cache_ttl, cache_clear=cache_clear, vision_deficiency=vision_deficiency, webhook=webhook, raise_on_upstream_error=raise_on_upstream_error )Create a ScreenshotConfig instance from a dictionary.
Methods
def to_api_params(self, key: str) ‑> Dict-
Expand source code
def to_api_params(self, key:str) -> Dict: params = { 'key': self.key or key, 'url': self.url } if self.format: params['format'] = Format(self.format).value if self.capture: params['capture'] = self.capture if self.resolution: params['resolution'] = self.resolution if self.country is not None: params['country'] = self.country if self.timeout is not None: params['timeout'] = self.timeout if self.rendering_wait is not None: params['rendering_wait'] = self.rendering_wait if self.wait_for_selector is not None: params['wait_for_selector'] = self.wait_for_selector if self.options is not None: params["options"] = ",".join(flag.value for flag in self.options) if self.auto_scroll is not None: params['auto_scroll'] = self._bool_to_http(self.auto_scroll) if self.js: params['js'] = base64.urlsafe_b64encode(self.js.encode('utf-8')).decode('utf-8') if self.cache is not None: params['cache'] = self._bool_to_http(self.cache) if self.cache_ttl is not None: params['cache_ttl'] = self.cache_ttl if self.cache_clear is not None: params['cache_clear'] = self._bool_to_http(self.cache_clear) else: if self.cache_ttl is not None: logging.warning('Params "cache_ttl" is ignored. Works only if cache is enabled') if self.cache_clear is not None: logging.warning('Params "cache_clear" is ignored. Works only if cache is enabled') if self.vision_deficiency is not None: params['vision_deficiency'] = self.vision_deficiency.value if self.webhook is not None: params['webhook_name'] = self.webhook return params def to_dict(self) ‑> Dict-
Expand source code
def to_dict(self) -> Dict: """ Export the ScreenshotConfig instance to a plain dictionary. """ return { 'url': self.url, 'format': Format(self.format).value if self.format else None, 'capture': self.capture, 'resolution': self.resolution, 'country': self.country, 'timeout': self.timeout, 'rendering_wait': self.rendering_wait, 'wait_for_selector': self.wait_for_selector, 'options': [Options(option).value for option in self.options] if self.options else None, 'auto_scroll': self.auto_scroll, 'js': self.js, 'cache': self.cache, 'cache_ttl': self.cache_ttl, 'cache_clear': self.cache_clear, 'vision_deficiency': self.vision_deficiency.value if self.vision_deficiency else None, 'webhook': self.webhook, 'raise_on_upstream_error': self.raise_on_upstream_error }Export the ScreenshotConfig instance to a plain dictionary.
class UpdateScheduleRequest (recurrence: ScheduleRecurrence | None = None,
scheduled_date: str | None = None,
allow_concurrency: bool | None = None,
retry_on_failure: bool | None = None,
max_retries: int | None = None,
notes: str | None = None,
scrape_config: Dict[str, Any] | None = None,
screenshot_config: Dict[str, Any] | None = None,
crawler_config: Dict[str, Any] | None = None)-
Expand source code
@dataclass class UpdateScheduleRequest: """Patch payload. Only fields with a non-None value are forwarded.""" recurrence: Optional[ScheduleRecurrence] = None scheduled_date: Optional[str] = None allow_concurrency: Optional[bool] = None retry_on_failure: Optional[bool] = None max_retries: Optional[int] = None notes: Optional[str] = None scrape_config: Optional[Dict[str, Any]] = None screenshot_config: Optional[Dict[str, Any]] = None crawler_config: Optional[Dict[str, Any]] = None def to_dict(self) -> Dict[str, Any]: out: Dict[str, Any] = {} if self.recurrence is not None: out["recurrence"] = self.recurrence.to_dict() if self.scheduled_date is not None: out["scheduled_date"] = self.scheduled_date if self.allow_concurrency is not None: out["allow_concurrency"] = self.allow_concurrency if self.retry_on_failure is not None: out["retry_on_failure"] = self.retry_on_failure if self.max_retries is not None: out["max_retries"] = self.max_retries if self.notes is not None: out["notes"] = self.notes if self.scrape_config is not None: out["scrape_config"] = self.scrape_config if self.screenshot_config is not None: out["screenshot_config"] = self.screenshot_config if self.crawler_config is not None: out["crawler_config"] = self.crawler_config return outPatch payload. Only fields with a non-None value are forwarded.
Instance variables
var allow_concurrency : bool | Nonevar crawler_config : Dict[str, Any] | Nonevar max_retries : int | Nonevar notes : str | Nonevar recurrence : ScheduleRecurrence | Nonevar retry_on_failure : bool | Nonevar scheduled_date : str | Nonevar scrape_config : Dict[str, Any] | Nonevar screenshot_config : Dict[str, Any] | None
Methods
def to_dict(self) ‑> Dict[str, Any]-
Expand source code
def to_dict(self) -> Dict[str, Any]: out: Dict[str, Any] = {} if self.recurrence is not None: out["recurrence"] = self.recurrence.to_dict() if self.scheduled_date is not None: out["scheduled_date"] = self.scheduled_date if self.allow_concurrency is not None: out["allow_concurrency"] = self.allow_concurrency if self.retry_on_failure is not None: out["retry_on_failure"] = self.retry_on_failure if self.max_retries is not None: out["max_retries"] = self.max_retries if self.notes is not None: out["notes"] = self.notes if self.scrape_config is not None: out["scrape_config"] = self.scrape_config if self.screenshot_config is not None: out["screenshot_config"] = self.screenshot_config if self.crawler_config is not None: out["crawler_config"] = self.crawler_config return out
class UpstreamHttpClientError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class UpstreamHttpClientError(UpstreamHttpError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.UpstreamHttpError
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
Subclasses
class UpstreamHttpError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class UpstreamHttpError(HttpError): passCommon base class for all non-exit exceptions.
Ancestors
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
Subclasses
class UpstreamHttpServerError (request: requests.models.Request,
response: requests.models.Response | None = None,
**kwargs)-
Expand source code
class UpstreamHttpServerError(UpstreamHttpClientError): passCommon base class for all non-exit exceptions.
Ancestors
- UpstreamHttpClientError
- scrapfly.errors.UpstreamHttpError
- scrapfly.errors.HttpError
- ScrapflyError
- builtins.Exception
- builtins.BaseException
class VisionDeficiency (value, names=None, *, module=None, qualname=None, type=None, start=1)-
Expand source code
class VisionDeficiency(Enum): """ Simulate vision deficiency for accessibility testing (WCAG compliance) Attributes: DEUTERANOPIA: Difficulty distinguishing green from red; green appears beige/gray PROTANOPIA: Reduced sensitivity to red light; red appears dark/black TRITANOPIA: Difficulty distinguishing blue from yellow and violet from red ACHROMATOPSIA: Complete inability to perceive color; sees only in grayscale REDUCED_CONTRAST: Simulates reduced contrast due to aging, low light, or other factors BLURRED_VISION: Simulates uncorrected refractive errors or age-related vision loss """ DEUTERANOPIA = "deuteranopia" PROTANOPIA = "protanopia" TRITANOPIA = "tritanopia" ACHROMATOPSIA = "achromatopsia" REDUCED_CONTRAST = "reducedContrast" BLURRED_VISION = "blurredVision"Simulate vision deficiency for accessibility testing (WCAG compliance)
Attributes
DEUTERANOPIA- Difficulty distinguishing green from red; green appears beige/gray
PROTANOPIA- Reduced sensitivity to red light; red appears dark/black
TRITANOPIA- Difficulty distinguishing blue from yellow and violet from red
ACHROMATOPSIA- Complete inability to perceive color; sees only in grayscale
REDUCED_CONTRAST- Simulates reduced contrast due to aging, low light, or other factors
BLURRED_VISION- Simulates uncorrected refractive errors or age-related vision loss
Ancestors
- enum.Enum
Class variables
var ACHROMATOPSIAvar BLURRED_VISIONvar DEUTERANOPIAvar PROTANOPIAvar REDUCED_CONTRASTvar TRITANOPIA
class WarcParser (warc_data: bytes |) -
Expand source code
class WarcParser: """ Parser for WARC files with automatic decompression Provides methods to iterate through WARC records and extract page data. Example: ```python # From bytes parser = WarcParser(warc_bytes) # Iterate all records for record in parser.iter_records(): print(f"{record.url}: {record.status_code}") # Get only HTTP responses for record in parser.iter_responses(): print(f"Page: {record.url}") html = record.content.decode('utf-8') # Get all pages as simple dicts pages = parser.get_pages() for page in pages: print(f"{page['url']}: {page['status_code']}") ``` """ def __init__(self, warc_data: Union[bytes, BinaryIO]): """ Initialize WARC parser Args: warc_data: WARC data as bytes or file-like object (supports both gzip-compressed and uncompressed) """ if isinstance(warc_data, bytes): # Try to decompress if gzipped if warc_data[:2] == b'\x1f\x8b': # gzip magic number try: warc_data = gzip.decompress(warc_data) except Exception: pass # Not gzipped or decompression failed self._data = BytesIO(warc_data) else: self._data = warc_data def iter_records(self) -> Iterator[WarcRecord]: """ Iterate through all WARC records Yields: WarcRecord: Each record in the WARC file """ self._data.seek(0) while True: # Read WARC version line version_line = self._read_line() if not version_line or not version_line.startswith(b'WARC/'): break # Read WARC headers warc_headers = self._read_headers() if not warc_headers: break # Get content length content_length = int(warc_headers.get('Content-Length', 0)) # Read content block content_block = self._data.read(content_length) # Skip trailing newlines self._read_line() self._read_line() # Parse the record record = self._parse_record(warc_headers, content_block) if record: yield record def iter_responses(self) -> Iterator[WarcRecord]: """ Iterate through HTTP response records only Filters out non-response records (requests, metadata, etc.) Yields: WarcRecord: HTTP response records only """ for record in self.iter_records(): if record.record_type == 'response' and record.status_code: yield record def get_pages(self) -> List[Dict]: """ Get all crawled pages as simple dictionaries This is the easiest way to access crawl results without dealing with WARC format details. Returns: List of dicts with keys: url, status_code, headers, content Example: ```python pages = parser.get_pages() for page in pages: print(f"{page['url']}: {len(page['content'])} bytes") html = page['content'].decode('utf-8') ``` """ pages = [] for record in self.iter_responses(): pages.append({ 'url': record.url, 'status_code': record.status_code, 'headers': record.headers, 'content': record.content }) return pages def _read_line(self) -> bytes: """Read a single line from the WARC file""" line = self._data.readline() return line.rstrip(b'\r\n') def _read_headers(self) -> Dict[str, str]: """Read headers until empty line""" headers = {} while True: line = self._read_line() if not line: break # Parse header line if b':' in line: key, value = line.split(b':', 1) headers[key.decode('utf-8').strip()] = value.decode('utf-8').strip() return headers def _parse_record(self, warc_headers: Dict[str, str], content_block: bytes) -> Optional[WarcRecord]: """Parse a WARC record from headers and content""" record_type = warc_headers.get('WARC-Type', '') url = warc_headers.get('WARC-Target-URI', '') if record_type == 'response': # Parse HTTP response http_headers, body = self._parse_http_response(content_block) status_code = self._extract_status_code(content_block) return WarcRecord( record_type=record_type, url=url, headers=http_headers, content=body, status_code=status_code, warc_headers=warc_headers ) elif record_type in ['request', 'metadata', 'warcinfo']: # Other record types - store raw content return WarcRecord( record_type=record_type, url=url, headers={}, content=content_block, status_code=None, warc_headers=warc_headers ) return None def _parse_http_response(self, content_block: bytes) -> tuple: """Parse HTTP response into headers and body""" try: # Split on double newline (end of headers) parts = content_block.split(b'\r\n\r\n', 1) if len(parts) < 2: parts = content_block.split(b'\n\n', 1) if len(parts) == 2: header_section, body = parts else: header_section, body = content_block, b'' # Parse headers headers = {} lines = header_section.split(b'\r\n') if b'\r\n' in header_section else header_section.split(b'\n') # Skip status line for line in lines[1:]: if b':' in line: key, value = line.split(b':', 1) headers[key.decode('utf-8', errors='ignore').strip()] = value.decode('utf-8', errors='ignore').strip() return headers, body except Exception: return {}, content_block def _extract_status_code(self, content_block: bytes) -> Optional[int]: """Extract HTTP status code from response""" try: # Look for HTTP status line (e.g., "HTTP/1.1 200 OK") first_line = content_block.split(b'\r\n', 1)[0] if b'\r\n' in content_block else content_block.split(b'\n', 1)[0] match = re.match(rb'HTTP/\d\.\d (\d+)', first_line) if match: return int(match.group(1)) except Exception: pass return NoneParser for WARC files with automatic decompression
Provides methods to iterate through WARC records and extract page data.
Example
# From bytes parser = WarcParser(warc_bytes) # Iterate all records for record in parser.iter_records(): print(f"{record.url}: {record.status_code}") # Get only HTTP responses for record in parser.iter_responses(): print(f"Page: {record.url}") html = record.content.decode('utf-8') # Get all pages as simple dicts pages = parser.get_pages() for page in pages: print(f"{page['url']}: {page['status_code']}")Initialize WARC parser
Args
warc_data- WARC data as bytes or file-like object (supports both gzip-compressed and uncompressed)
Methods
def get_pages(self) ‑> List[Dict]-
Expand source code
def get_pages(self) -> List[Dict]: """ Get all crawled pages as simple dictionaries This is the easiest way to access crawl results without dealing with WARC format details. Returns: List of dicts with keys: url, status_code, headers, content Example: ```python pages = parser.get_pages() for page in pages: print(f"{page['url']}: {len(page['content'])} bytes") html = page['content'].decode('utf-8') ``` """ pages = [] for record in self.iter_responses(): pages.append({ 'url': record.url, 'status_code': record.status_code, 'headers': record.headers, 'content': record.content }) return pagesGet all crawled pages as simple dictionaries
This is the easiest way to access crawl results without dealing with WARC format details.
Returns
Listofdicts with keys- url, status_code, headers, content
Example
pages = parser.get_pages() for page in pages: print(f"{page['url']}: {len(page['content'])} bytes") html = page['content'].decode('utf-8') def iter_records(self) ‑> Iterator[WarcRecord]-
Expand source code
def iter_records(self) -> Iterator[WarcRecord]: """ Iterate through all WARC records Yields: WarcRecord: Each record in the WARC file """ self._data.seek(0) while True: # Read WARC version line version_line = self._read_line() if not version_line or not version_line.startswith(b'WARC/'): break # Read WARC headers warc_headers = self._read_headers() if not warc_headers: break # Get content length content_length = int(warc_headers.get('Content-Length', 0)) # Read content block content_block = self._data.read(content_length) # Skip trailing newlines self._read_line() self._read_line() # Parse the record record = self._parse_record(warc_headers, content_block) if record: yield record def iter_responses(self) ‑> Iterator[WarcRecord]-
Expand source code
def iter_responses(self) -> Iterator[WarcRecord]: """ Iterate through HTTP response records only Filters out non-response records (requests, metadata, etc.) Yields: WarcRecord: HTTP response records only """ for record in self.iter_records(): if record.record_type == 'response' and record.status_code: yield recordIterate through HTTP response records only
Filters out non-response records (requests, metadata, etc.)
Yields
WarcRecord- HTTP response records only
class WarcRecord (record_type: str,
url: str,
headers: Dict[str, str],
content: bytes,
status_code: int | None,
warc_headers: Dict[str, str])-
Expand source code
@dataclass class WarcRecord: """ Represents a single WARC record A WARC file contains multiple records, each representing a captured HTTP transaction or metadata. """ record_type: str # Type of record (response, request, metadata, etc.) url: str # Associated URL headers: Dict[str, str] # HTTP headers content: bytes # Response body/content status_code: Optional[int] # HTTP status code (for response records) warc_headers: Dict[str, str] # WARC-specific headers def __repr__(self): return f"WarcRecord(type={self.record_type}, url={self.url}, status={self.status_code})"Represents a single WARC record
A WARC file contains multiple records, each representing a captured HTTP transaction or metadata.
Instance variables
var content : bytesvar headers : Dict[str, str]var record_type : strvar status_code : int | Nonevar url : strvar warc_headers : Dict[str, str]