curl --request POST \
--url https://api.prefetch.io/crawl \
--header 'Content-Type: application/json' \
--header 'X-API-Key: <api-key>' \
--data '
{
"url": "https://docs.stripe.com",
"limit": 25,
"max_depth": 2,
"include_paths": [
"^/docs/"
],
"exclude_paths": [
"^/docs/changelog/"
],
"allow_subdomains": false,
"allow_external_links": false,
"ignore_query_params": true,
"respect_robots_txt": true,
"use_sitemap": true,
"delay_ms": 0,
"concurrency": 2,
"scrape_options": {
"formats": [
"markdown"
],
"only_main_content": true,
"include_links": true,
"include_images": true,
"include_base64_images": false,
"include_selectors": [
"<string>"
],
"exclude_selectors": [
"<string>"
],
"json_prompt": "<string>",
"json_schema": {},
"summary_max_words": 120
}
}
'import requests
url = "https://api.prefetch.io/crawl"
payload = {
"url": "https://docs.stripe.com",
"limit": 25,
"max_depth": 2,
"include_paths": ["^/docs/"],
"exclude_paths": ["^/docs/changelog/"],
"allow_subdomains": False,
"allow_external_links": False,
"ignore_query_params": True,
"respect_robots_txt": True,
"use_sitemap": True,
"delay_ms": 0,
"concurrency": 2,
"scrape_options": {
"formats": ["markdown"],
"only_main_content": True,
"include_links": True,
"include_images": True,
"include_base64_images": False,
"include_selectors": ["<string>"],
"exclude_selectors": ["<string>"],
"json_prompt": "<string>",
"json_schema": {},
"summary_max_words": 120
}
}
headers = {
"X-API-Key": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'X-API-Key': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
url: 'https://docs.stripe.com',
limit: 25,
max_depth: 2,
include_paths: ['^/docs/'],
exclude_paths: ['^/docs/changelog/'],
allow_subdomains: false,
allow_external_links: false,
ignore_query_params: true,
respect_robots_txt: true,
use_sitemap: true,
delay_ms: 0,
concurrency: 2,
scrape_options: {
formats: ['markdown'],
only_main_content: true,
include_links: true,
include_images: true,
include_base64_images: false,
include_selectors: ['<string>'],
exclude_selectors: ['<string>'],
json_prompt: '<string>',
json_schema: {},
summary_max_words: 120
}
})
};
fetch('https://api.prefetch.io/crawl', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.prefetch.io/crawl",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'url' => 'https://docs.stripe.com',
'limit' => 25,
'max_depth' => 2,
'include_paths' => [
'^/docs/'
],
'exclude_paths' => [
'^/docs/changelog/'
],
'allow_subdomains' => false,
'allow_external_links' => false,
'ignore_query_params' => true,
'respect_robots_txt' => true,
'use_sitemap' => true,
'delay_ms' => 0,
'concurrency' => 2,
'scrape_options' => [
'formats' => [
'markdown'
],
'only_main_content' => true,
'include_links' => true,
'include_images' => true,
'include_base64_images' => false,
'include_selectors' => [
'<string>'
],
'exclude_selectors' => [
'<string>'
],
'json_prompt' => '<string>',
'json_schema' => [
],
'summary_max_words' => 120
]
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"X-API-Key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.prefetch.io/crawl"
payload := strings.NewReader("{\n \"url\": \"https://docs.stripe.com\",\n \"limit\": 25,\n \"max_depth\": 2,\n \"include_paths\": [\n \"^/docs/\"\n ],\n \"exclude_paths\": [\n \"^/docs/changelog/\"\n ],\n \"allow_subdomains\": false,\n \"allow_external_links\": false,\n \"ignore_query_params\": true,\n \"respect_robots_txt\": true,\n \"use_sitemap\": true,\n \"delay_ms\": 0,\n \"concurrency\": 2,\n \"scrape_options\": {\n \"formats\": [\n \"markdown\"\n ],\n \"only_main_content\": true,\n \"include_links\": true,\n \"include_images\": true,\n \"include_base64_images\": false,\n \"include_selectors\": [\n \"<string>\"\n ],\n \"exclude_selectors\": [\n \"<string>\"\n ],\n \"json_prompt\": \"<string>\",\n \"json_schema\": {},\n \"summary_max_words\": 120\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("X-API-Key", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.prefetch.io/crawl")
.header("X-API-Key", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"url\": \"https://docs.stripe.com\",\n \"limit\": 25,\n \"max_depth\": 2,\n \"include_paths\": [\n \"^/docs/\"\n ],\n \"exclude_paths\": [\n \"^/docs/changelog/\"\n ],\n \"allow_subdomains\": false,\n \"allow_external_links\": false,\n \"ignore_query_params\": true,\n \"respect_robots_txt\": true,\n \"use_sitemap\": true,\n \"delay_ms\": 0,\n \"concurrency\": 2,\n \"scrape_options\": {\n \"formats\": [\n \"markdown\"\n ],\n \"only_main_content\": true,\n \"include_links\": true,\n \"include_images\": true,\n \"include_base64_images\": false,\n \"include_selectors\": [\n \"<string>\"\n ],\n \"exclude_selectors\": [\n \"<string>\"\n ],\n \"json_prompt\": \"<string>\",\n \"json_schema\": {},\n \"summary_max_words\": 120\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.prefetch.io/crawl")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["X-API-Key"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"url\": \"https://docs.stripe.com\",\n \"limit\": 25,\n \"max_depth\": 2,\n \"include_paths\": [\n \"^/docs/\"\n ],\n \"exclude_paths\": [\n \"^/docs/changelog/\"\n ],\n \"allow_subdomains\": false,\n \"allow_external_links\": false,\n \"ignore_query_params\": true,\n \"respect_robots_txt\": true,\n \"use_sitemap\": true,\n \"delay_ms\": 0,\n \"concurrency\": 2,\n \"scrape_options\": {\n \"formats\": [\n \"markdown\"\n ],\n \"only_main_content\": true,\n \"include_links\": true,\n \"include_images\": true,\n \"include_base64_images\": false,\n \"include_selectors\": [\n \"<string>\"\n ],\n \"exclude_selectors\": [\n \"<string>\"\n ],\n \"json_prompt\": \"<string>\",\n \"json_schema\": {},\n \"summary_max_words\": 120\n }\n}"
response = http.request(request)
puts response.read_body{
"success": true,
"data": {
"id": "crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e",
"status": "pending",
"url": "https://docs.stripe.com",
"limit": 50,
"max_depth": 2,
"created_at": "2026-08-20T09:14:03.221Z",
"expires_at": "2026-08-21T09:14:03.221Z"
},
"meta": {
"requestId": "c5f4e3d2-9b8c-4d5e-af3a-2b4c6d8e0f21",
"durationMs": 84
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}POST /crawl
Crawl a whole site and get every page back as clean markdown. Costs 3 credits to submit, plus 3 per page returned.
curl --request POST \
--url https://api.prefetch.io/crawl \
--header 'Content-Type: application/json' \
--header 'X-API-Key: <api-key>' \
--data '
{
"url": "https://docs.stripe.com",
"limit": 25,
"max_depth": 2,
"include_paths": [
"^/docs/"
],
"exclude_paths": [
"^/docs/changelog/"
],
"allow_subdomains": false,
"allow_external_links": false,
"ignore_query_params": true,
"respect_robots_txt": true,
"use_sitemap": true,
"delay_ms": 0,
"concurrency": 2,
"scrape_options": {
"formats": [
"markdown"
],
"only_main_content": true,
"include_links": true,
"include_images": true,
"include_base64_images": false,
"include_selectors": [
"<string>"
],
"exclude_selectors": [
"<string>"
],
"json_prompt": "<string>",
"json_schema": {},
"summary_max_words": 120
}
}
'import requests
url = "https://api.prefetch.io/crawl"
payload = {
"url": "https://docs.stripe.com",
"limit": 25,
"max_depth": 2,
"include_paths": ["^/docs/"],
"exclude_paths": ["^/docs/changelog/"],
"allow_subdomains": False,
"allow_external_links": False,
"ignore_query_params": True,
"respect_robots_txt": True,
"use_sitemap": True,
"delay_ms": 0,
"concurrency": 2,
"scrape_options": {
"formats": ["markdown"],
"only_main_content": True,
"include_links": True,
"include_images": True,
"include_base64_images": False,
"include_selectors": ["<string>"],
"exclude_selectors": ["<string>"],
"json_prompt": "<string>",
"json_schema": {},
"summary_max_words": 120
}
}
headers = {
"X-API-Key": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'X-API-Key': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
url: 'https://docs.stripe.com',
limit: 25,
max_depth: 2,
include_paths: ['^/docs/'],
exclude_paths: ['^/docs/changelog/'],
allow_subdomains: false,
allow_external_links: false,
ignore_query_params: true,
respect_robots_txt: true,
use_sitemap: true,
delay_ms: 0,
concurrency: 2,
scrape_options: {
formats: ['markdown'],
only_main_content: true,
include_links: true,
include_images: true,
include_base64_images: false,
include_selectors: ['<string>'],
exclude_selectors: ['<string>'],
json_prompt: '<string>',
json_schema: {},
summary_max_words: 120
}
})
};
fetch('https://api.prefetch.io/crawl', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.prefetch.io/crawl",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'url' => 'https://docs.stripe.com',
'limit' => 25,
'max_depth' => 2,
'include_paths' => [
'^/docs/'
],
'exclude_paths' => [
'^/docs/changelog/'
],
'allow_subdomains' => false,
'allow_external_links' => false,
'ignore_query_params' => true,
'respect_robots_txt' => true,
'use_sitemap' => true,
'delay_ms' => 0,
'concurrency' => 2,
'scrape_options' => [
'formats' => [
'markdown'
],
'only_main_content' => true,
'include_links' => true,
'include_images' => true,
'include_base64_images' => false,
'include_selectors' => [
'<string>'
],
'exclude_selectors' => [
'<string>'
],
'json_prompt' => '<string>',
'json_schema' => [
],
'summary_max_words' => 120
]
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"X-API-Key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.prefetch.io/crawl"
payload := strings.NewReader("{\n \"url\": \"https://docs.stripe.com\",\n \"limit\": 25,\n \"max_depth\": 2,\n \"include_paths\": [\n \"^/docs/\"\n ],\n \"exclude_paths\": [\n \"^/docs/changelog/\"\n ],\n \"allow_subdomains\": false,\n \"allow_external_links\": false,\n \"ignore_query_params\": true,\n \"respect_robots_txt\": true,\n \"use_sitemap\": true,\n \"delay_ms\": 0,\n \"concurrency\": 2,\n \"scrape_options\": {\n \"formats\": [\n \"markdown\"\n ],\n \"only_main_content\": true,\n \"include_links\": true,\n \"include_images\": true,\n \"include_base64_images\": false,\n \"include_selectors\": [\n \"<string>\"\n ],\n \"exclude_selectors\": [\n \"<string>\"\n ],\n \"json_prompt\": \"<string>\",\n \"json_schema\": {},\n \"summary_max_words\": 120\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("X-API-Key", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.prefetch.io/crawl")
.header("X-API-Key", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"url\": \"https://docs.stripe.com\",\n \"limit\": 25,\n \"max_depth\": 2,\n \"include_paths\": [\n \"^/docs/\"\n ],\n \"exclude_paths\": [\n \"^/docs/changelog/\"\n ],\n \"allow_subdomains\": false,\n \"allow_external_links\": false,\n \"ignore_query_params\": true,\n \"respect_robots_txt\": true,\n \"use_sitemap\": true,\n \"delay_ms\": 0,\n \"concurrency\": 2,\n \"scrape_options\": {\n \"formats\": [\n \"markdown\"\n ],\n \"only_main_content\": true,\n \"include_links\": true,\n \"include_images\": true,\n \"include_base64_images\": false,\n \"include_selectors\": [\n \"<string>\"\n ],\n \"exclude_selectors\": [\n \"<string>\"\n ],\n \"json_prompt\": \"<string>\",\n \"json_schema\": {},\n \"summary_max_words\": 120\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.prefetch.io/crawl")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["X-API-Key"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"url\": \"https://docs.stripe.com\",\n \"limit\": 25,\n \"max_depth\": 2,\n \"include_paths\": [\n \"^/docs/\"\n ],\n \"exclude_paths\": [\n \"^/docs/changelog/\"\n ],\n \"allow_subdomains\": false,\n \"allow_external_links\": false,\n \"ignore_query_params\": true,\n \"respect_robots_txt\": true,\n \"use_sitemap\": true,\n \"delay_ms\": 0,\n \"concurrency\": 2,\n \"scrape_options\": {\n \"formats\": [\n \"markdown\"\n ],\n \"only_main_content\": true,\n \"include_links\": true,\n \"include_images\": true,\n \"include_base64_images\": false,\n \"include_selectors\": [\n \"<string>\"\n ],\n \"exclude_selectors\": [\n \"<string>\"\n ],\n \"json_prompt\": \"<string>\",\n \"json_schema\": {},\n \"summary_max_words\": 120\n }\n}"
response = http.request(request)
puts response.read_body{
"success": true,
"data": {
"id": "crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e",
"status": "pending",
"url": "https://docs.stripe.com",
"limit": 50,
"max_depth": 2,
"created_at": "2026-08-20T09:14:03.221Z",
"expires_at": "2026-08-21T09:14:03.221Z"
},
"meta": {
"requestId": "c5f4e3d2-9b8c-4d5e-af3a-2b4c6d8e0f21",
"durationMs": 84
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}Overview
/crawl walks a site breadth-first from a start URL and renders every page it finds, exactly the way GET /scrape would render it.
Crawls run for minutes, so this is the one asynchronous endpoint in the API. You submit a crawl, get an id back straight away, then poll GET /crawl/{id} for progress and results.
Submit
POST /crawl returns 202 with a crawl id.Poll
GET /crawl/{id} returns progress plus a page of results. Follow next until it is null.Stop early (optional)
DELETE /crawl/{id} cancels a running crawl. Pages already fetched stay readable.Example request
curl -X POST "https://api.prefetch.io/crawl" \
-H "X-API-Key: $PREFETCH_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"url": "https://docs.stripe.com",
"limit": 50,
"max_depth": 2,
"include_paths": ["^/docs/"],
"exclude_paths": ["^/docs/changelog/"],
"scrape_options": { "formats": ["markdown"] }
}'
const res = await fetch("https://api.prefetch.io/crawl", {
method: "POST",
headers: {
"X-API-Key": process.env.PREFETCH_API_KEY,
"Content-Type": "application/json",
},
body: JSON.stringify({
url: "https://docs.stripe.com",
limit: 50,
max_depth: 2,
include_paths: ["^/docs/"],
scrape_options: { formats: ["markdown"] },
}),
});
const { data } = await res.json();
console.log(data.id); // crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e
import requests, os
r = requests.post(
"https://api.prefetch.io/crawl",
headers={"X-API-Key": os.environ["PREFETCH_API_KEY"]},
json={
"url": "https://docs.stripe.com",
"limit": 50,
"max_depth": 2,
"include_paths": ["^/docs/"],
"scrape_options": {"formats": ["markdown"]},
},
)
crawl_id = r.json()["data"]["id"]
Example response
{
"success": true,
"data": {
"id": "crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e",
"status": "pending",
"url": "https://docs.stripe.com",
"limit": 50,
"max_depth": 2,
"created_at": "2026-08-20T09:14:03.221Z",
"expires_at": "2026-08-21T09:14:03.221Z"
},
"meta": {
"requestId": "c5f4e3d2-9b8c-4d5e-af3a-2b4c6d8e0f21",
"durationMs": 84
}
}
202 Accepted, not 200 — the crawl has been queued, not completed.
Controlling scope
A crawl stays inside the boundary you draw. By default that means the site you pointed it at, two links deep, at most 25 pages.| Parameter | Default | What it does |
|---|---|---|
limit | 25 | Maximum pages fetched. Hard ceiling: 500. |
max_depth | 2 | How many links deep to follow. Hard ceiling: 5. |
include_paths | — | Regular expressions. A URL’s path must match at least one. |
exclude_paths | — | Regular expressions. A matching URL is skipped. |
allow_subdomains | false | Follow links onto subdomains. |
allow_external_links | false | Follow links onto other sites. |
^/docs/ means the docs section, and a pattern containing the hostname will never match. Exclusions win over inclusions. An invalid regular expression is rejected at submission with a 400, not silently ignored halfway through a crawl.
{
"url": "https://docs.stripe.com",
"include_paths": ["^/docs/payments/"],
"exclude_paths": ["^/docs/payments/legacy/", "\\?locale="]
}
Sitemap seeding and depth
Withuse_sitemap on (the default), the crawl seeds itself from the site’s sitemap as well as following links.
Sitemap-seeded pages sit at depth 0 — the site handed them over, you did not follow a link to reach them. That makes one particularly useful combination:
{ "url": "https://docs.stripe.com", "max_depth": 0, "use_sitemap": true, "limit": 200 }
Politeness
Crawls obeyrobots.txt by default — both Disallow rules and Crawl-delay. Set respect_robots_txt: false to ignore the rules.
robots.txt being read. Its Sitemap: entries always seed the crawl, because a sitemap tells you where a site’s pages are whether or not you are honouring its restrictions. Use use_sitemap: false to skip seeding.delay_ms adds a pause between pages, and concurrency (1–5, default 2) sets how many are fetched in parallel. A Crawl-delay in robots.txt wins whenever it asks for more space than delay_ms.
Rendering each page
scrape_options takes the same options as GET /scrape, so a crawled page and a scraped page are the same object.
{
"url": "https://docs.stripe.com",
"scrape_options": {
"formats": ["markdown", "links"],
"only_main_content": true,
"exclude_selectors": [".sidebar", ".version-banner"]
}
}
summary or json on a 500-page crawl means 500 LLM calls. Start with a small limit to check the output before scaling up.Credits
You are charged 3 credits to submit, plus 3 credits per page the crawl returns. Pages are billed as they complete, on the status endpoint, so:- You pay for pages actually fetched, never for the
limityou asked for. - Pages that failed (
ok: false) are not charged. - Polling repeatedly never charges twice for the same page.
- A cancelled crawl is charged only for what it fetched before stopping.
Next
GET /crawl/{id}
DELETE /crawl/{id}
Authorizations
Your Prefetch API key. Obtain one from the dashboard.
Body
Where the crawl starts.
"https://docs.stripe.com"
Maximum pages to crawl.
1 <= x <= 500How many links deep to follow from the start URL. Sitemap-seeded pages sit at depth 0, so max_depth: 0 with use_sitemap: true crawls exactly what the sitemap lists and follows nothing.
0 <= x <= 5Regular expressions matched against a URL's path and query. A URL must match at least one to be crawled.
["^/docs/"]
Regular expressions. A URL matching any of them is skipped. Exclusions win over inclusions.
["^/docs/changelog/"]
Follow links onto subdomains of the start URL.
Follow links onto other sites.
Treat URLs that differ only by query string as one page.
Obey the rules in robots.txt — Disallow and Crawl-delay. Sitemap discovery happens either way.
Seed the crawl from the site's sitemap as well as from links.
Politeness delay between pages. A larger Crawl-delay in robots.txt wins.
0 <= x <= 30000Pages fetched in parallel.
1 <= x <= 5How each crawled page is rendered. The same options GET /scrape takes.
Show child attributes
Show child attributes