{"openapi":"3.1.0","info":{"title":"CrawlFox API","version":"1.0.0","summary":"Search and scrape the web in one API call.","description":"The CrawlFox REST API. Scrape a URL to clean, LLM-ready content, search Google or DuckDuckGo, batch-scrape, map a site's URLs, crawl a site, extract structured fields with CSS selectors, and read back your own request logs.\n\nUsage is metered in credits, not requests: there is no per-minute rate limit, so you don't need backoff for one. Failed calls are free.","contact":{"name":"CrawlFox support","email":"support@crawlfox.io","url":"https://crawlfox.io/docs"},"termsOfService":"https://crawlfox.io/legal/terms"},"servers":[{"url":"https://api.crawlfox.io","description":"Production"}],"tags":[{"name":"Scrape","description":"Fetch pages as clean, structured content."},{"name":"Search","description":"Ranked organic results from Google or DuckDuckGo."},{"name":"Crawl","description":"Map a site's URLs, or crawl it and scrape every page."},{"name":"Logs","description":"Read back your own past requests and their results."},{"name":"Service","description":"Unauthenticated health probes."}],"security":[{"bearerAuth":[]}],"paths":{"/v1/scrape":{"post":{"operationId":"scrape","summary":"Scrape a URL","description":"Fetch a single URL and return clean, structured data. Blocked and JavaScript-heavy pages are handled for you. Costs 1 credit per page regardless of how many formats you request; a page served from cache costs the same as a fresh one, and failed calls are free.","tags":["Scrape"],"security":[{"bearerAuth":[]}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"$ref":"#/components/schemas/ScrapeRequest"}}}},"responses":{"200":{"description":"The scraped page.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ScrapeResponse"}}}},"400":{"description":"The url is missing or not fully-qualified.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"401":{"description":"Missing or invalid API key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"429":{"description":"Rate limited, or out of credits. Read `code` to tell them apart: MONTHLY_QUOTA_EXCEEDED means the balance is exhausted and needs topping up, RATE_LIMIT_EXCEEDED and CONCURRENCY_LIMIT_EXCEEDED mean the key is going too fast and the request can be retried.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"500":{"description":"Unexpected error on our side. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"502":{"description":"The target site could not be retrieved.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"503":{"description":"Temporarily unable to serve the request — authentication is unavailable, or the fetch never reached the site. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"504":{"description":"The target site did not respond in time. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}}}}},"/v1/scrape/{url}":{"get":{"operationId":"scrapeUrl","summary":"Scrape a URL (GET)","description":"Markdown-only scrape of one URL, with no request body — the cheapest read path. No links, images or emails sidecars; use POST /v1/scrape when you need other formats. Costs 1 credit.","tags":["Scrape"],"security":[{"bearerAuth":[]}],"parameters":[{"name":"url","in":"path","required":true,"description":"The URL to scrape, percent-encoded.","schema":{"type":"string"}}],"responses":{"200":{"description":"The scraped page, markdown only.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ScrapeResponse"}}}},"400":{"description":"The url is missing or not fully-qualified.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"401":{"description":"Missing or invalid API key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"429":{"description":"Rate limited, or out of credits. Read `code` to tell them apart: MONTHLY_QUOTA_EXCEEDED means the balance is exhausted and needs topping up, RATE_LIMIT_EXCEEDED and CONCURRENCY_LIMIT_EXCEEDED mean the key is going too fast and the request can be retried.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"500":{"description":"Unexpected error on our side. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"502":{"description":"The target site could not be retrieved.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"503":{"description":"Temporarily unable to serve the request — authentication is unavailable, or the fetch never reached the site. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"504":{"description":"The target site did not respond in time. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}}}}},"/v1/batch":{"post":{"operationId":"batchScrape","summary":"Scrape many URLs","description":"Scrape up to 100 URLs in one call. Each is processed independently with the same options, and results come back in input order. Billed per URL at the single-scrape rate; failed URLs are free.","tags":["Scrape"],"security":[{"bearerAuth":[]}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"$ref":"#/components/schemas/BatchScrapeRequest"}}}},"responses":{"200":{"description":"One result per input URL, in order.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/BatchScrapeResponse"}}}},"400":{"description":"The urls array is missing or empty.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"401":{"description":"Missing or invalid API key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"429":{"description":"Rate limited, or out of credits. Read `code` to tell them apart: MONTHLY_QUOTA_EXCEEDED means the balance is exhausted and needs topping up, RATE_LIMIT_EXCEEDED and CONCURRENCY_LIMIT_EXCEEDED mean the key is going too fast and the request can be retried.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"500":{"description":"Unexpected error on our side. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"503":{"description":"Temporarily unable to serve the request — authentication is unavailable, or the fetch never reached the site. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}}}}},"/v1/search":{"post":{"operationId":"search","summary":"Search the web","description":"Run a search on Google, Bing or DuckDuckGo and get ranked organic results. Choose the engine with the engine field in the request body; omit it for Google. An ?engine= query parameter is also accepted, and the body wins when both are set. Costs 1 credit per 10 requested results, charged on the requested count whether or not that many are found.","tags":["Search"],"security":[{"bearerAuth":[]}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"$ref":"#/components/schemas/SerpRequest"}}}},"responses":{"200":{"description":"Ranked organic results.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/SerpResponse"}}}},"400":{"description":"The query is missing, or the engine is not recognised.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"401":{"description":"Missing or invalid API key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"429":{"description":"Rate limited, or out of credits. Read `code` to tell them apart: MONTHLY_QUOTA_EXCEEDED means the balance is exhausted and needs topping up, RATE_LIMIT_EXCEEDED and CONCURRENCY_LIMIT_EXCEEDED mean the key is going too fast and the request can be retried.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"500":{"description":"Unexpected error on our side. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"502":{"description":"The engine could not be reached.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"503":{"description":"Temporarily unable to serve the request — authentication is unavailable, or the fetch never reached the site. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"504":{"description":"The engine did not respond in time. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}}}}},"/v1/search/stream":{"post":{"operationId":"searchStream","summary":"Search the web (streamed)","description":"Same search, streamed as newline-delimited JSON so you can render results as they land instead of waiting for the full set. Worth using when num is above 10: a page is 10 results, so a larger request spans several pages and each arrives as it lands. At num 10 or below the search is a single page and this endpoint sends one page then done, no sooner than the non-streaming one returns. Each line is an object with a type of \"page\", \"done\" or \"error\"; a \"page\" carries a data.web array, and \"done\" carries partial, creditsUsed and id. Because the stream commits HTTP 200 before the work finishes, upstream failures arrive as an in-band \"error\" line, not as an HTTP status.","tags":["Search"],"security":[{"bearerAuth":[]}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"$ref":"#/components/schemas/SerpRequest"}}}},"responses":{"200":{"description":"NDJSON event stream — one JSON object per line.","content":{"application/x-ndjson":{"schema":{"type":"string"}}}},"400":{"description":"The query is missing, or the engine is not recognised.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"401":{"description":"Missing or invalid API key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"429":{"description":"Rate limited, or out of credits. Read `code` to tell them apart: MONTHLY_QUOTA_EXCEEDED means the balance is exhausted and needs topping up, RATE_LIMIT_EXCEEDED and CONCURRENCY_LIMIT_EXCEEDED mean the key is going too fast and the request can be retried.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"500":{"description":"Unexpected error on our side. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"503":{"description":"Temporarily unable to serve the request — authentication is unavailable, or the fetch never reached the site. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}}}}},"/v1/logs/{id}":{"get":{"operationId":"getLog","summary":"Look up a past request","description":"Status and timing of one of your own past requests. Free. A request that isn't yours is indistinguishable from one that doesn't exist — both return LOG_NOT_FOUND.","tags":["Logs"],"security":[{"bearerAuth":[]}],"parameters":[{"name":"id","in":"path","required":true,"description":"Request id, as returned in metadata.scrapeId or the SERP response id.","schema":{"type":"string","format":"uuid"}}],"responses":{"200":{"description":"The request log.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/LogRow"}}}},"401":{"description":"Missing or invalid API key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"404":{"description":"No such request, or it doesn't belong to this key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"429":{"description":"Rate limited, or out of credits. Read `code` to tell them apart: MONTHLY_QUOTA_EXCEEDED means the balance is exhausted and needs topping up, RATE_LIMIT_EXCEEDED and CONCURRENCY_LIMIT_EXCEEDED mean the key is going too fast and the request can be retried.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"500":{"description":"Unexpected error on our side. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"503":{"description":"Temporarily unable to serve the request — authentication is unavailable, or the fetch never reached the site. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}}}}},"/v1/logs/{id}/result":{"get":{"operationId":"getLogResult","summary":"Retrieve a past result","description":"The stored response body of one of your own past scrapes, returned verbatim. Free. Results are retained only while storage is configured and within its retention window; when a result isn't available the response is LOG_NOT_FOUND, the same as for an unknown id.","tags":["Logs"],"security":[{"bearerAuth":[]}],"parameters":[{"name":"id","in":"path","required":true,"description":"Request id, as returned in metadata.scrapeId or the SERP response id.","schema":{"type":"string","format":"uuid"}}],"responses":{"200":{"description":"The stored result body.","content":{"application/json":{"schema":{"type":"object","additionalProperties":true}}}},"401":{"description":"Missing or invalid API key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"404":{"description":"No such request, no retained result, or it isn't yours.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"429":{"description":"Rate limited, or out of credits. Read `code` to tell them apart: MONTHLY_QUOTA_EXCEEDED means the balance is exhausted and needs topping up, RATE_LIMIT_EXCEEDED and CONCURRENCY_LIMIT_EXCEEDED mean the key is going too fast and the request can be retried.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"500":{"description":"Unexpected error on our side. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"503":{"description":"Temporarily unable to serve the request — authentication is unavailable, or the fetch never reached the site. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}}}}},"/v1/map":{"post":{"operationId":"map","summary":"Map a site's URLs","description":"Return the URLs a site has — its sitemap plus the links on the start page — without scraping them. Costs 1 credit per call.","tags":["Crawl"],"security":[{"bearerAuth":[]}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"$ref":"#/components/schemas/MapRequest"}}}},"responses":{"200":{"description":"The site's URLs.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/MapResponse"}}}},"400":{"description":"The url is missing or not publicly reachable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"401":{"description":"Missing or invalid API key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"429":{"description":"Rate limited, or out of credits. Read `code` to tell them apart: MONTHLY_QUOTA_EXCEEDED means the balance is exhausted and needs topping up, RATE_LIMIT_EXCEEDED and CONCURRENCY_LIMIT_EXCEEDED mean the key is going too fast and the request can be retried.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"500":{"description":"Unexpected error on our side. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"503":{"description":"Temporarily unable to serve the request — authentication is unavailable, or the fetch never reached the site. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}}}}},"/v1/crawl":{"post":{"operationId":"crawl","summary":"Crawl a site","description":"Start a crawl from a URL: its links are followed and every page found is scraped. Returns the job id at once; poll GET /v1/crawl/{id} until status is completed. Costs 1 credit per page per body format; failed pages are free.","tags":["Crawl"],"security":[{"bearerAuth":[]}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"$ref":"#/components/schemas/CrawlRequest"}}}},"responses":{"202":{"description":"The crawl was started.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/CrawlCreateResponse"}}}},"400":{"description":"The url is missing, not publicly reachable, disallowed by robots.txt (pass ignoreRobotsTxt), or a selector in scrapeOptions does not parse.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"401":{"description":"Missing or invalid API key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"429":{"description":"Rate limited, or out of credits. Read `code` to tell them apart: MONTHLY_QUOTA_EXCEEDED means the balance is exhausted and needs topping up, RATE_LIMIT_EXCEEDED and CONCURRENCY_LIMIT_EXCEEDED mean the key is going too fast and the request can be retried.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"500":{"description":"Unexpected error on our side. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"503":{"description":"Temporarily unable to serve the request — authentication is unavailable, or the fetch never reached the site. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}}}}},"/v1/crawl/{id}":{"get":{"operationId":"crawlStatus","summary":"Read a crawl","description":"Status, counts and the pages scraped so far. Large results are paged: follow next (which carries ?skip=) until it is absent. Free.","tags":["Crawl"],"security":[{"bearerAuth":[]}],"parameters":[{"name":"id","in":"path","required":true,"description":"The crawl id returned by POST /v1/crawl.","schema":{"type":"string","format":"uuid"}},{"name":"skip","in":"query","required":false,"description":"Pages to skip.","schema":{"type":"integer","minimum":0}},{"name":"limit","in":"query","required":false,"description":"Pages per read, up to 100.","schema":{"type":"integer","minimum":1,"maximum":100}}],"responses":{"200":{"description":"The crawl's status and a page of its documents.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/CrawlStatusResponse"}}}},"401":{"description":"Missing or invalid API key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"404":{"description":"No such crawl, or it isn't yours.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"429":{"description":"Rate limited, or out of credits. Read `code` to tell them apart: MONTHLY_QUOTA_EXCEEDED means the balance is exhausted and needs topping up, RATE_LIMIT_EXCEEDED and CONCURRENCY_LIMIT_EXCEEDED mean the key is going too fast and the request can be retried.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"500":{"description":"Unexpected error on our side. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"503":{"description":"Temporarily unable to serve the request — authentication is unavailable, or the fetch never reached the site. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}}}},"delete":{"operationId":"crawlCancel","summary":"Cancel a crawl","description":"Stop a running crawl. Pages already scraped stay readable. Free.","tags":["Crawl"],"security":[{"bearerAuth":[]}],"parameters":[{"name":"id","in":"path","required":true,"description":"The crawl id returned by POST /v1/crawl.","schema":{"type":"string","format":"uuid"}}],"responses":{"200":{"description":"Cancelled.","content":{"application/json":{"schema":{"type":"object","properties":{"status":{"type":"string","enum":["cancelled"]}}}}}},"401":{"description":"Missing or invalid API key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"404":{"description":"No such crawl, or it isn't yours.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"429":{"description":"Rate limited, or out of credits. Read `code` to tell them apart: MONTHLY_QUOTA_EXCEEDED means the balance is exhausted and needs topping up, RATE_LIMIT_EXCEEDED and CONCURRENCY_LIMIT_EXCEEDED mean the key is going too fast and the request can be retried.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"500":{"description":"Unexpected error on our side. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"503":{"description":"Temporarily unable to serve the request — authentication is unavailable, or the fetch never reached the site. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}}}}},"/v1/crawl/{id}/errors":{"get":{"operationId":"crawlErrors","summary":"A crawl's failed pages","description":"The pages a crawl could not scrape, with the reason, and the URLs robots.txt kept it from. Free.","tags":["Crawl"],"security":[{"bearerAuth":[]}],"parameters":[{"name":"id","in":"path","required":true,"description":"The crawl id returned by POST /v1/crawl.","schema":{"type":"string","format":"uuid"}}],"responses":{"200":{"description":"The failed pages.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/CrawlErrorsResponse"}}}},"401":{"description":"Missing or invalid API key.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"404":{"description":"No such crawl, or it isn't yours.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"429":{"description":"Rate limited, or out of credits. Read `code` to tell them apart: MONTHLY_QUOTA_EXCEEDED means the balance is exhausted and needs topping up, RATE_LIMIT_EXCEEDED and CONCURRENCY_LIMIT_EXCEEDED mean the key is going too fast and the request can be retried.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"500":{"description":"Unexpected error on our side. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}},"503":{"description":"Temporarily unable to serve the request — authentication is unavailable, or the fetch never reached the site. Retryable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/Error"}}}}}}},"/health":{"get":{"operationId":"health","summary":"Service health","description":"Readiness probe. No authentication, no credits.","tags":["Service"],"security":[],"responses":{"200":{"description":"The service is healthy.","content":{"application/json":{"schema":{"type":"object","properties":{"status":{"type":"string"},"version":{"type":"string"}}}}}}}}}},"components":{"securitySchemes":{"bearerAuth":{"type":"http","scheme":"bearer","description":"Your secret API key, sent as `Authorization: Bearer <key>`. Create and reveal keys from the dashboard."}},"schemas":{"ScrapeRequest":{"type":"object","required":["url"],"properties":{"url":{"type":"string","format":"uri","description":"The URL to scrape. Must be fully-qualified (include the scheme)."},"formats":{"type":"array","items":{"type":"string","enum":["markdown","html","rawHtml","json","links","images","emails"]},"default":["markdown"],"description":"Output formats to return. Requesting several still costs 1 credit."},"skipCache":{"type":"boolean","default":false,"description":"Force a fresh fetch instead of serving a cached result."},"zdr":{"type":"boolean","default":false,"description":"Zero data retention. When true, the page is not written to cache or durable storage, the URL is omitted from logs, and skipCache is forced. The response is returned once and not retained server-side. Still costs 1 credit per page."},"country":{"type":"string","description":"ISO 3166-1 alpha-2 country code (e.g. us, de) for geo-pinned egress and Accept-Language. When country, language, and location are all omitted, the request uses us/en internally without forcing geo egress; an explicit value routes through geo-capable proxies."},"language":{"type":"string","description":"Two-letter language code (e.g. en, de) for the Accept-Language header. Defaults to en when omitted."},"timeout":{"type":"integer","default":90000,"minimum":1000,"description":"How long to wait for the page, in milliseconds. Defaults to 90000. Below 1000 the request is rejected; a value above 90000 is capped at 90000 rather than refused."},"jsonOptions":{"type":"object","required":["selectors"],"description":"Deterministic, AI-free structured extraction. Applied when \"json\" is among formats: the listed CSS selectors run against the fetched DOM and the result comes back as data.json. This — not a separate endpoint — is how extraction works.","properties":{"selectors":{"type":"object","description":"Output key → selector rule.","additionalProperties":{"oneOf":[{"type":"string","description":"Shorthand — first match's text content."},{"type":"object","required":["selector"],"properties":{"selector":{"type":"string","description":"CSS selector to match."},"attr":{"type":"string","description":"Which part of the match to read. Omitted or \"text\" gives collapsed text; \"html\" gives inner HTML; anything else is read as an attribute name."},"multiple":{"type":"boolean","default":false,"description":"Return every match as an array instead of the first match."}}}]}}}},"location":{"type":"object","description":"Firecrawl-compatible nested alias for country and language. Flat country and language fields win when both are set.","properties":{"country":{"type":"string","description":"ISO 3166-1 alpha-2 country code (e.g. US, de). Normalised to lowercase."},"languages":{"type":"array","items":{"type":"string"},"description":"Priority-ordered language list; only the first entry is used."}}},"redactPII":{"oneOf":[{"type":"boolean","description":"When true, redact PII from markdown, html, and json output using the default entity set."},{"type":"object","properties":{"mode":{"type":"string","enum":["accurate","aggressive","fast"],"description":"Redaction strategy. Defaults to accurate."},"entities":{"type":"array","items":{"type":"string","enum":["PERSON","EMAIL","PHONE","LOCATION","FINANCIAL","SECRET"]},"description":"Which PII types to redact. Omitted means all supported types."},"replaceStyle":{"type":"string","enum":["tag","mask","remove"],"description":"How matched spans are replaced. Defaults to tag (<EMAIL>, etc.)."}}}],"description":"Redact personally identifiable information from markdown, html, and json before returning. Cannot be combined with the emails format."},"parsePDF":{"type":"boolean","default":true,"description":"When the URL is a PDF, extract its text as markdown. When false a PDF is refused with 415 UNSUPPORTED_CONTENT_TYPE and not charged."},"includeTags":{"type":"array","items":{"type":"string"},"maxItems":50,"description":"CSS selectors for the only parts of the page to keep, such as article or #main. This is applied before the page is converted, so every format reflects it. A selector that does not parse returns a 400."},"excludeTags":{"type":"array","items":{"type":"string"},"maxItems":50,"description":"CSS selectors for the parts of the page to drop, such as nav or .cookie-banner. These are applied after includeTags."}}},"BatchScrapeRequest":{"type":"object","required":["urls"],"properties":{"urls":{"type":"array","items":{"type":"string","format":"uri"},"minItems":1,"description":"URLs to scrape — up to 100 per request."},"formats":{"type":"array","items":{"type":"string","enum":["markdown","html","rawHtml","json","links","images","emails"]},"default":["markdown"],"description":"Output formats to return. Requesting several still costs 1 credit."},"skipCache":{"type":"boolean","default":false,"description":"Force a fresh fetch instead of serving a cached result."},"zdr":{"type":"boolean","default":false,"description":"Zero data retention. When true, the page is not written to cache or durable storage, the URL is omitted from logs, and skipCache is forced. The response is returned once and not retained server-side. Still costs 1 credit per page."},"country":{"type":"string","description":"ISO 3166-1 alpha-2 country code (e.g. us, de) for geo-pinned egress and Accept-Language. When country, language, and location are all omitted, the request uses us/en internally without forcing geo egress; an explicit value routes through geo-capable proxies."},"language":{"type":"string","description":"Two-letter language code (e.g. en, de) for the Accept-Language header. Defaults to en when omitted."},"timeout":{"type":"integer","default":90000,"minimum":1000,"description":"How long to wait for the page, in milliseconds. Defaults to 90000. Below 1000 the request is rejected; a value above 90000 is capped at 90000 rather than refused. Applies to each URL on its own, not to the batch as a whole — a batch of slow pages can take longer than this in total."},"location":{"type":"object","description":"Firecrawl-compatible nested alias for country and language. Flat country and language fields win when both are set.","properties":{"country":{"type":"string","description":"ISO 3166-1 alpha-2 country code (e.g. US, de). Normalised to lowercase."},"languages":{"type":"array","items":{"type":"string"},"description":"Priority-ordered language list; only the first entry is used."}}},"redactPII":{"oneOf":[{"type":"boolean","description":"When true, redact PII from markdown, html, and json output using the default entity set."},{"type":"object","properties":{"mode":{"type":"string","enum":["accurate","aggressive","fast"],"description":"Redaction strategy. Defaults to accurate."},"entities":{"type":"array","items":{"type":"string","enum":["PERSON","EMAIL","PHONE","LOCATION","FINANCIAL","SECRET"]},"description":"Which PII types to redact. Omitted means all supported types."},"replaceStyle":{"type":"string","enum":["tag","mask","remove"],"description":"How matched spans are replaced. Defaults to tag (<EMAIL>, etc.)."}}}],"description":"Redact personally identifiable information from markdown, html, and json before returning. Cannot be combined with the emails format."}}},"PageMetadata":{"type":"object","required":["sourceURL","url","statusCode","scrapeId","creditsUsed","cacheState"],"properties":{"title":{"type":"string"},"description":{"type":"string"},"language":{"type":"string"},"sourceURL":{"type":"string","description":"The URL as requested."},"url":{"type":"string","description":"The final URL after redirects."},"statusCode":{"type":"integer","description":"HTTP status returned by the target."},"contentType":{"type":"string"},"favicon":{"type":"string"},"numPages":{"type":"integer","description":"Number of pages in the document. Present only for PDF sources; omitted for HTML."},"totalPages":{"type":"integer","description":"Total pages in the document. Present only for PDF sources; omitted for HTML. Equal to numPages for a whole-document scrape."},"scrapeId":{"type":"string","description":"Id of this run — use it with GET /v1/logs/{id}."},"creditsUsed":{"type":"integer"},"cacheState":{"type":"string","description":"Whether the result was served from cache."}}},"ScrapeData":{"type":"object","required":["metadata"],"description":"Only the formats you asked for are present; the rest are omitted.","properties":{"markdown":{"type":"string"},"html":{"type":"string"},"rawHtml":{"type":"string"},"links":{"type":"array","items":{"type":"string"}},"images":{"type":"array","items":{"type":"string"}},"emails":{"type":"array","items":{"type":"string"}},"json":{"type":"object","description":"Result of the jsonOptions selectors.","additionalProperties":true},"metadata":{"$ref":"#/components/schemas/PageMetadata"}}},"ScrapeResponse":{"type":"object","required":["success","data"],"properties":{"success":{"type":"boolean"},"data":{"$ref":"#/components/schemas/ScrapeData"}}},"BatchScrapeResponse":{"type":"object","required":["success","count","results"],"description":"Results are in input order. A URL that failed is still an object in results, with success: false, so the array stays uniformly typed.","properties":{"success":{"type":"boolean"},"count":{"type":"integer","description":"Number of entries in results."},"results":{"type":"array","items":{"$ref":"#/components/schemas/ScrapeResponse"}}}},"SerpRequest":{"type":"object","required":["q"],"properties":{"engine":{"type":"string","enum":["google","bing","duckduckgo"],"default":"google","description":"Search engine to query. Omit or leave blank for Google. Also accepted as an ?engine= query parameter; this field wins when both are set."},"q":{"type":"string","description":"The search query."},"num":{"type":"integer","default":10,"maximum":100,"description":"Results to return. Hard-capped at 100 across every engine."},"start":{"type":"integer","default":0,"description":"Zero-based offset for pagination."},"country":{"type":"string","default":"us","description":"Two-letter country code."},"language":{"type":"string","default":"en","description":"Two-letter language code."}}},"SerpResult":{"type":"object","required":["url","title","description","position"],"properties":{"url":{"type":"string"},"title":{"type":"string"},"description":{"type":"string"},"position":{"type":"integer","description":"1-based rank within the returned set."}}},"SerpResponse":{"type":"object","required":["success","data","creditsUsed","id"],"properties":{"success":{"type":"boolean"},"data":{"type":"object","required":["web"],"properties":{"web":{"type":"array","items":{"$ref":"#/components/schemas/SerpResult"}}}},"creditsUsed":{"type":"integer"},"id":{"type":"string","description":"Id of this request — use it with GET /v1/logs/{id}."}}},"LogRow":{"type":"object","required":["id","started_at_ms","duration_ms","url","status"],"description":"Customer projection of a request log. Internal delivery detail is not included.","properties":{"id":{"type":"string"},"started_at_ms":{"type":"integer","format":"int64","description":"Start time, epoch milliseconds."},"duration_ms":{"type":"integer"},"url":{"description":"The URL from the original request."},"status":{"type":"integer","description":"Terminal HTTP status of the request."},"final_outcome":{"type":"string","nullable":true}}},"Error":{"type":"object","required":["code","status","retryable","title","message","remediation"],"description":"Switch on code — statuses and copy may change, codes will not. Retryable errors are safe to retry with backoff.","properties":{"code":{"type":"string","enum":["MISSING_URL","INVALID_URL","BAD_REQUEST","MISSING_API_KEY","INVALID_API_KEY","AUTH_UNAVAILABLE","MONTHLY_QUOTA_EXCEEDED","RATE_LIMIT_EXCEEDED","CONCURRENCY_LIMIT_EXCEEDED","UNKNOWN_SERP_ENGINE","SERP_ENGINE_UNAVAILABLE","FORMAT_UNAVAILABLE","LOG_NOT_FOUND","RESULT_UNAVAILABLE","REQUEST_TIMEOUT","FETCH_FAILED","FETCH_TIMEOUT","UPSTREAM_UNREACHABLE","UPSTREAM_NOT_FOUND","UPSTREAM_FORBIDDEN","UPSTREAM_RATE_LIMITED","UPSTREAM_SERVER_ERROR","UPSTREAM_TIMEOUT","UPSTREAM_GEO_BLOCKED","BOT_WALL","NO_PUBLIC_CONTENT","EXTRACTION_FAILED","UNSUPPORTED_CONTENT_TYPE","SCANNED_PDF_NO_TEXT","RENDER_CAPACITY_UNAVAILABLE","INTERNAL_ERROR"]},"status":{"type":"integer"},"retryable":{"type":"boolean"},"title":{"type":"string"},"message":{"type":"string"},"remediation":{"type":"string"}}},"MapRequest":{"type":"object","required":["url"],"properties":{"url":{"type":"string","format":"uri","description":"The site, or a section of it, to map."},"search":{"type":"string","description":"Only return URLs containing this text."},"limit":{"type":"integer","minimum":1,"default":5000,"description":"The most URLs to return."},"includeSubdomains":{"type":"boolean","default":true},"ignoreQueryParameters":{"type":"boolean","default":true,"description":"Treat URLs differing only by query string as one."},"sitemap":{"type":"string","enum":["include","only","skip"],"default":"include","description":"include reads the sitemap and the page; only reads just the sitemap; skip ignores it."}}},"MapResponse":{"type":"object","required":["success","links"],"properties":{"success":{"type":"boolean"},"id":{"type":"string","format":"uuid"},"links":{"type":"array","items":{"type":"object","required":["url"],"properties":{"url":{"type":"string","format":"uri"},"title":{"type":"string"},"description":{"type":"string"}}}}}},"CrawlRequest":{"type":"object","required":["url"],"properties":{"url":{"type":"string","format":"uri","description":"Where the crawl starts. Stays under this path unless crawlEntireDomain is true."},"limit":{"type":"integer","minimum":1,"default":10000,"description":"The most pages to scrape. Values above 100000 are capped."},"maxDiscoveryDepth":{"type":"integer","minimum":0,"description":"Link hops from the start page. Omit for no limit."},"includePaths":{"type":"array","items":{"type":"string"},"description":"Regexes; only matching paths are crawled."},"excludePaths":{"type":"array","items":{"type":"string"},"description":"Regexes; matching paths are skipped."},"crawlEntireDomain":{"type":"boolean","default":false},"allowSubdomains":{"type":"boolean","default":false},"allowExternalLinks":{"type":"boolean","default":false},"ignoreQueryParameters":{"type":"boolean","default":false},"regexOnFullURL":{"type":"boolean","default":false,"description":"Match includePaths/excludePaths against the whole URL, not just the path."},"ignoreRobotsTxt":{"type":"boolean","default":true},"sitemap":{"type":"string","enum":["include","only","skip"],"default":"include","description":"include reads the sitemap and the page; only reads just the sitemap; skip ignores it."},"delay":{"type":"number","minimum":0,"maximum":60,"description":"Seconds between pages of this crawl."},"maxConcurrency":{"type":"integer","minimum":1,"maximum":10,"default":4},"scrapeOptions":{"type":"object","description":"How each page is scraped. The same options as POST /v1/scrape.","properties":{"formats":{"type":"array","items":{"type":"string","enum":["markdown","html","rawHtml","json","links","images","emails"]},"default":["markdown"]},"parsePDF":{"type":"boolean","default":true,"description":"When the URL is a PDF, extract its text as markdown. When false a PDF is refused with 415 UNSUPPORTED_CONTENT_TYPE and not charged."},"includeTags":{"type":"array","items":{"type":"string"},"maxItems":50,"description":"CSS selectors for the only parts of the page to keep, such as article or #main. This is applied before the page is converted, so every format reflects it. A selector that does not parse returns a 400."},"excludeTags":{"type":"array","items":{"type":"string"},"maxItems":50,"description":"CSS selectors for the parts of the page to drop, such as nav or .cookie-banner. These are applied after includeTags."}}},"webhook":{"type":"object","required":["url"],"properties":{"url":{"type":"string","format":"uri","description":"Called when the crawl finishes."}}}}},"CrawlCreateResponse":{"type":"object","required":["success","id","url"],"properties":{"success":{"type":"boolean"},"id":{"type":"string","format":"uuid","description":"The crawl id. Poll GET /v1/crawl/{id} with it."},"url":{"type":"string","format":"uri","description":"The status URL for this crawl."}}},"CrawlDocument":{"type":"object","required":["metadata"],"description":"One crawled page, in the same shape as a scrape's data.","properties":{"markdown":{"type":"string"},"html":{"type":"string"},"rawHtml":{"type":"string"},"json":{"description":"Structured data, when json was among the formats."},"links":{"type":"array","items":{"type":"string"}},"images":{"type":"array","items":{"type":"string"}},"metadata":{"$ref":"#/components/schemas/PageMetadata"}}},"CrawlStatusResponse":{"type":"object","required":["success","status","total","completed","creditsUsed","data"],"properties":{"success":{"type":"boolean"},"status":{"type":"string","enum":["scraping","completed","cancelled","failed"]},"total":{"type":"integer","description":"Pages discovered so far."},"completed":{"type":"integer","description":"Pages finished (scraped or failed)."},"creditsUsed":{"type":"number"},"expiresAt":{"type":"string","format":"date-time","description":"When the stored pages are dropped."},"next":{"type":"string","format":"uri","description":"The next page of data, when there is one (carries ?skip=)."},"data":{"type":"array","items":{"$ref":"#/components/schemas/CrawlDocument"}}}},"CrawlErrorsResponse":{"type":"object","required":["errors","robotsBlocked"],"properties":{"errors":{"type":"array","items":{"type":"object","properties":{"id":{"type":"string"},"timestamp":{"type":"string"},"url":{"type":"string"},"error":{"type":"string"}}}},"robotsBlocked":{"type":"array","items":{"type":"string"},"description":"URLs robots.txt kept the crawl from fetching."}}}}}}