payload.url | string | Yes | — | The starting URL for the crawl (must include http:// or https:// protocol) Format: uri. |
payload.maxPages | integer | No | 100 | Maximum number of pages to crawl. Hard cap: 500. Minimum: 1. Maximum: 500. |
payload.maxDepth | integer | No | — | Maximum link depth from the starting URL (0 = only the starting page) Minimum: 0. |
payload.urlRegex | string | No | — | Regex pattern. Only URLs matching this pattern will be followed and scraped. |
payload.includeLinks | boolean | No | true | Preserve hyperlinks in the Markdown output |
payload.includeImages | boolean | No | false | Include image references in the Markdown output |
payload.shortenBase64Images | boolean | No | true | Truncate base64-encoded image data in the Markdown output |
payload.useMainContentOnly | boolean | No | false | Extract only the main content, stripping headers, footers, sidebars, and navigation |
payload.followSubdomains | boolean | No | false | When true, follow links on subdomains of the starting URL’s domain (e.g. docs.example.com when starting from example.com). www and apex are always treated as equivalent. |
payload.pdf | object | No | {"shouldParse":true,"ocr":false} | PDF parsing controls. Use start/end to limit text extraction and embedded-image detection/OCR to an inclusive 1-based page range. |
payload.includeFrames | boolean | No | false | When true, the contents of iframes are rendered to Markdown for each crawled page. |
payload.includeSelectors | array | No | — | CSS selectors. When provided, only matching HTML subtrees (and their descendants) are kept before each crawled page is converted to Markdown. When omitted, the entire document is kept. Examples: “article.main”, “#content”, “[role=main]”. Maximum items: 50. |
payload.excludeSelectors | array | No | — | CSS selectors to remove before each crawled page is converted to Markdown. Applied after includeSelectors. Exclusion takes precedence: an element matching both is removed. Examples: “nav”, “footer”, “.ad-banner”, “[aria-hidden=true]”. Maximum items: 50. |
payload.maxAgeMs | integer | No | 86400000 | Return a cached result if a prior scrape for the same parameters exists and is younger than this many milliseconds. Defaults to 1 day (86400000 ms) when omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. Minimum: 0. Maximum: 2592000000. |
payload.waitForMs | integer | No | — | Optional browser wait time in milliseconds after initial page load for each crawled page. Min: 0. Max: 30000 (30 seconds). Minimum: 0. Maximum: 30000. |
payload.settleAnimations | boolean | No | false | When true, waits briefly for CSS and transition animations to settle before extracting each crawled page. Defaults to false. This adds a bit of latency in exchange for more stable output on animated pages. |
payload.stopAfterMs | integer | No | 80000 | Soft time budget for the crawl in milliseconds. After each scrape, the crawler checks the elapsed time and, if exceeded, returns the pages collected so far instead of continuing. Min: 10000 (10s). Max: 110000 (110s). Default: 80000 (80s). Minimum: 10000. Maximum: 110000. |
payload.country | "ad" | "ae" | "af" | "ag" | "ai" | "al" | "am" | "ao" | "ar" | "at" | "au" | "aw" | "az" | "ba" | "bb" | "bd" | "be" | "bf" | "bg" | "bh" | "bi" | "bj" | "bm" | "bn" | "bo" | "bq" | "br" | "bs" | "bw" | "by" | "bz" | "ca" | "cd" | "cf" | "cg" | "ch" | "ci" | "cl" | "cm" | "cn" | "co" | "cr" | "cv" | "cw" | "cy" | "cz" | "de" | "dj" | "dk" | "dm" | "do" | "dz" | "ec" | "ee" | "eg" | "es" | "et" | "fi" | "fj" | "fr" | "ga" | "gb" | "gd" | "ge" | "gf" | "gg" | "gh" | "gm" | "gn" | "gp" | "gq" | "gr" | "gt" | "gu" | "gw" | "gy" | "hk" | "hn" | "hr" | "ht" | "hu" | "id" | "ie" | "il" | "im" | "in" | "iq" | "ir" | "is" | "it" | "je" | "jm" | "jo" | "jp" | "ke" | "kg" | "kh" | "kn" | "kr" | "kw" | "ky" | "kz" | "la" | "lb" | "lc" | "lk" | "lr" | "ls" | "lt" | "lu" | "lv" | "ly" | "ma" | "mc" | "md" | "me" | "mf" | "mg" | "mk" | "ml" | "mm" | "mn" | "mo" | "mq" | "mr" | "mt" | "mu" | "mv" | "mw" | "mx" | "my" | "mz" | "na" | "nc" | "ne" | "ng" | "ni" | "nl" | "no" | "np" | "nz" | "om" | "pa" | "pe" | "pf" | "pg" | "ph" | "pk" | "pl" | "pr" | "ps" | "pt" | "py" | "qa" | "re" | "ro" | "rs" | "ru" | "rw" | "sa" | "sc" | "sd" | "se" | "sg" | "si" | "sk" | "sl" | "sm" | "sn" | "so" | "sr" | "ss" | "st" | "sv" | "sx" | "sy" | "sz" | "tc" | "td" | "tg" | "th" | "tj" | "tl" | "tm" | "tn" | "tr" | "tt" | "tw" | "tz" | "ua" | "ug" | "us" | "uy" | "uz" | "vc" | "ve" | "vg" | "vi" | "vn" | "ye" | "yt" | "za" | "zm" | "zw" | No | — | Fetch the target page through a residential proxy in this country (ISO 3166-1 alpha-2). Allowed: ad, ae, af, ag, ai, al, am, ao, ar, at, au, aw, az, ba, bb, bd, be, bf, bg, bh, bi, bj, bm, bn, bo, bq, br, bs, bw, by, bz, ca, cd, cf, cg, ch, ci, cl, cm, cn, co, cr, cv, cw, cy, cz, de, dj, dk, dm, do, dz, ec, ee, eg, es, et, fi, fj, fr, ga, gb, gd, ge, gf, gg, gh, See the live schema for the complete constraint. |
payload.timeoutMS | integer | No | — | Optional timeout in milliseconds for the request. If the request takes longer than this value, it will be aborted with a 408 status code. Maximum allowed value is 300000ms (5 minutes). Minimum: 1000. Maximum: 300000. |
payload.zdr | "enabled" | "disabled" | No | "disabled" | Set to enabled to bypass shared caches and omit request and response content from retained usage logs. Requires zero data retention to be enabled for your organization (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED. Successful ZDR responses include X-Context-ZDR: true. Allowed: enabled, disabled. |
payload.tags | array | No | — | Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters. Maximum items: 20. |