const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({url: 'https://example.com', maxPages: 10})
};
fetch('https://api.context.dev/v1/web/crawl', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));import requests
url = "https://api.context.dev/v1/web/crawl"
payload = {
"url": "https://example.com",
"maxPages": 10
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)require 'uri'
require 'net/http'
url = URI("https://api.context.dev/v1/web/crawl")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"url\": \"https://example.com\",\n \"maxPages\": 10\n}"
response = http.request(request)
puts response.read_bodypackage main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.context.dev/v1/web/crawl"
payload := strings.NewReader("{\n \"url\": \"https://example.com\",\n \"maxPages\": 10\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.context.dev/v1/web/crawl",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'url' => 'https://example.com',
'maxPages' => 10
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}curl --request POST \
--url https://api.context.dev/v1/web/crawl \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"url": "https://example.com",
"maxPages": 10
}
'{
"results": [
{
"markdown": "<string>",
"metadata": {
"sourceUrl": "<string>",
"finalUrl": "<string>",
"title": "<string>",
"url": "<string>",
"crawlDepth": 123,
"statusCode": 123,
"success": true,
"description": "<string>",
"language": "<string>",
"keywords": [
"<string>"
],
"canonicalUrl": "<string>",
"author": "<string>",
"siteName": "<string>",
"image": "<string>",
"favicon": "<string>",
"publishedTime": "<string>",
"modifiedTime": "<string>",
"robots": "<string>",
"openGraph": {},
"twitter": {},
"alternates": [
{
"href": "<string>",
"hreflang": "<string>",
"type": "<string>",
"title": "<string>"
}
],
"headings": [
{
"level": 3,
"text": "<string>"
}
],
"jsonLd": [
{}
],
"additionalMeta": {}
}
}
],
"metadata": {
"numUrls": 123,
"maxCrawlDepth": 123,
"numSucceeded": 123,
"numFailed": 123,
"numSkipped": 123
},
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"cache_metadata": {
"status": "hit",
"age_ms": 1
},
"partial": true,
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"message": "<string>",
"error_code": "INPUT_VALIDATION_ERROR",
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"message": "<string>",
"error_code": "INTERNAL_ERROR",
"required_permission": "logs:read",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"message": "<string>",
"error_code": "INTERNAL_ERROR",
"required_permission": "logs:read",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"message": "<string>",
"error_code": "NOT_FOUND",
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"message": "<string>",
"error_code": "REQUEST_TIMEOUT",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"message": "<string>",
"error_code": "UNSUPPORTED_CONTENT",
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"message": "<string>",
"error_code": "RATE_LIMITED",
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"message": "<string>",
"error_code": "INTERNAL_ERROR",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}Crawl a website
Crawl a website and return page content as Markdown. Use a batch for crawls beyond 500 pages.
const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({url: 'https://example.com', maxPages: 10})
};
fetch('https://api.context.dev/v1/web/crawl', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));import requests
url = "https://api.context.dev/v1/web/crawl"
payload = {
"url": "https://example.com",
"maxPages": 10
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)require 'uri'
require 'net/http'
url = URI("https://api.context.dev/v1/web/crawl")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"url\": \"https://example.com\",\n \"maxPages\": 10\n}"
response = http.request(request)
puts response.read_bodypackage main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.context.dev/v1/web/crawl"
payload := strings.NewReader("{\n \"url\": \"https://example.com\",\n \"maxPages\": 10\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.context.dev/v1/web/crawl",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'url' => 'https://example.com',
'maxPages' => 10
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}curl --request POST \
--url https://api.context.dev/v1/web/crawl \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"url": "https://example.com",
"maxPages": 10
}
'{
"results": [
{
"markdown": "<string>",
"metadata": {
"sourceUrl": "<string>",
"finalUrl": "<string>",
"title": "<string>",
"url": "<string>",
"crawlDepth": 123,
"statusCode": 123,
"success": true,
"description": "<string>",
"language": "<string>",
"keywords": [
"<string>"
],
"canonicalUrl": "<string>",
"author": "<string>",
"siteName": "<string>",
"image": "<string>",
"favicon": "<string>",
"publishedTime": "<string>",
"modifiedTime": "<string>",
"robots": "<string>",
"openGraph": {},
"twitter": {},
"alternates": [
{
"href": "<string>",
"hreflang": "<string>",
"type": "<string>",
"title": "<string>"
}
],
"headings": [
{
"level": 3,
"text": "<string>"
}
],
"jsonLd": [
{}
],
"additionalMeta": {}
}
}
],
"metadata": {
"numUrls": 123,
"maxCrawlDepth": 123,
"numSucceeded": 123,
"numFailed": 123,
"numSkipped": 123
},
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"cache_metadata": {
"status": "hit",
"age_ms": 1
},
"partial": true,
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"message": "<string>",
"error_code": "INPUT_VALIDATION_ERROR",
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"message": "<string>",
"error_code": "INTERNAL_ERROR",
"required_permission": "logs:read",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"message": "<string>",
"error_code": "INTERNAL_ERROR",
"required_permission": "logs:read",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"message": "<string>",
"error_code": "NOT_FOUND",
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"message": "<string>",
"error_code": "REQUEST_TIMEOUT",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"message": "<string>",
"error_code": "UNSUPPORTED_CONTENT",
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"message": "<string>",
"error_code": "RATE_LIMITED",
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}{
"request_id": "3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91",
"message": "<string>",
"error_code": "INTERNAL_ERROR",
"key_metadata": {
"credits_consumed": 123,
"credits_remaining": 123
}
}Authorizations
Send Authorization: Bearer <API_KEY>. Keys have full access unless restricted to scopes.
Body
Start URL, including http:// or https://.
Maximum pages to crawl.
1 <= x <= 500Maximum link depth from the starting URL (0 = only the starting page)
x >= 0Regex pattern. Only URLs matching this pattern will be followed and scraped. An automatic prefix scope in the form ^ follows a redirect of the starting page.
"^https?://[^/]+/blog/"
Preserve hyperlinks in the Markdown output
Include image references in the Markdown output
Truncate base64-encoded image data in the Markdown output
Extract only the main content, stripping headers, footers, sidebars, and navigation
When true, follow links on subdomains of the starting URL's domain (e.g. docs.example.com when starting from example.com). www and apex are always treated as equivalent.
PDF handling. start/end limit parsing to an inclusive, 1-based page range.
Show child attributes
Show child attributes
When true, the contents of iframes are rendered to Markdown for each crawled page.
Keep matching HTML subtrees before converting each page to Markdown.
502048Remove matching elements after inclusions. Exclusions take precedence.
502048Maximum cache age in milliseconds. Defaults to 1 day; 0 fetches fresh.
0 <= x <= 2592000000Browser wait time in milliseconds after initial page load for each crawled page. Defaults to 3500 (3.5 seconds). Min: 0. Max: 30000 (30 seconds).
0 <= x <= 30000Wait briefly for CSS animations and transitions to settle before reading each page.
Soft crawl deadline in milliseconds. Returns pages collected before the next deadline check.
10000 <= x <= 110000Fetch from this country (ISO 3166-1 alpha-2).
ad, ae, af, ag, ai, al, am, ao, ar, at, au, aw, az, ba, bb, bd, be, bf, bg, bh, bi, bj, bm, bn, bo, bq, br, bs, bw, by, bz, ca, cd, cf, cg, ch, ci, cl, cm, cn, co, cr, cv, cw, cy, cz, de, dj, dk, dm, do, dz, ec, ee, eg, es, et, fi, fj, fr, ga, gb, gd, ge, gf, gg, gh, gm, gn, gp, gq, gr, gt, gu, gw, gy, hk, hn, hr, ht, hu, id, ie, il, im, in, iq, ir, is, it, je, jm, jo, jp, ke, kg, kh, kn, kr, kw, ky, kz, la, lb, lc, lk, lr, ls, lt, lu, lv, ly, ma, mc, md, me, mf, mg, mk, ml, mm, mn, mo, mq, mr, mt, mu, mv, mw, mx, my, mz, na, nc, ne, ng, ni, nl, no, np, nz, om, pa, pe, pf, pg, ph, pk, pl, pr, ps, pt, py, qa, re, ro, rs, ru, rw, sa, sc, sd, se, sg, si, sk, sl, sm, sn, so, sr, ss, st, sv, sx, sy, sz, tc, td, tg, th, tj, tl, tm, tn, tr, tt, tw, tz, ua, ug, us, uy, uz, vc, ve, vg, vi, vn, ye, yt, za, zm, zw "de"
Request deadline and what to return when it passes.
Show child attributes
Show child attributes
enabled turns on zero data retention. Returns 403 ZDR_NOT_ENABLED unless your organization has ZDR.
enabled, disabled Labels for filtering usage in the dashboard.
201 - 50["production", "team-alpha"]
Response
Successful response
Show child attributes
Show child attributes
Show child attributes
Show child attributes
Unique ID of this request, also in X-Request-Id. Include it when contacting support.
"3f1c2a6e-8b4d-4c1e-9f0a-2d7b5e6c8a91"
Whether this response came from cache.
Show child attributes
Show child attributes
True when timeoutOpts.behavior=return-partial returned the usable results collected before the deadline. Partial collections are not cached as complete results.
Credits this request used and your remaining balance.
Show child attributes
Show child attributes
Was this page helpful?