Get crawl status and results
curl --request GET \
--url https://api.prefetch.io/crawl/{id} \
--header 'X-API-Key: <api-key>'import requests
url = "https://api.prefetch.io/crawl/{id}"
headers = {"X-API-Key": "<api-key>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {'X-API-Key': '<api-key>'}};
fetch('https://api.prefetch.io/crawl/{id}', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.prefetch.io/crawl/{id}",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"X-API-Key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.prefetch.io/crawl/{id}"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("X-API-Key", "<api-key>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.prefetch.io/crawl/{id}")
.header("X-API-Key", "<api-key>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.prefetch.io/crawl/{id}")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["X-API-Key"] = '<api-key>'
response = http.request(request)
puts response.read_body{
"success": true,
"data": {
"id": "crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e",
"status": "running",
"url": "https://docs.stripe.com",
"total": 38,
"completed": 12,
"failed": 1,
"credits_used": 36,
"created_at": "2026-08-20T09:14:03.221Z",
"started_at": "2026-08-20T09:14:04.010Z",
"finished_at": null,
"expires_at": "2026-08-21T09:14:03.221Z",
"error": null,
"next": "/crawl/crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e?skip=25&limit=25",
"data": [
{
"url": "https://docs.stripe.com/payments",
"depth": 0,
"status_code": 200,
"ok": true,
"error": null,
"final_url": "https://docs.stripe.com/payments",
"domain": "docs.stripe.com",
"metadata": {
"title": "Payments",
"description": "Accept payments online.",
"language": "en"
},
"markdown": {
"content": "# Payments\n\nAccept payments online, in person, and around the world...",
"char_count": 2841,
"word_count": 402,
"truncated": false,
"main_content_only": true
}
}
]
},
"meta": {
"requestId": "d6a5f4e3-ac9d-4e6f-b04b-3c5d7e9f1a32",
"durationMs": 96
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}Web extraction
GET /crawl/{id}
Poll a crawl for progress and read its pages. Charges 3 credits for each page completed since your last call.
GET
/
crawl
/
{id}
Get crawl status and results
curl --request GET \
--url https://api.prefetch.io/crawl/{id} \
--header 'X-API-Key: <api-key>'import requests
url = "https://api.prefetch.io/crawl/{id}"
headers = {"X-API-Key": "<api-key>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {'X-API-Key': '<api-key>'}};
fetch('https://api.prefetch.io/crawl/{id}', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.prefetch.io/crawl/{id}",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"X-API-Key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.prefetch.io/crawl/{id}"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("X-API-Key", "<api-key>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.prefetch.io/crawl/{id}")
.header("X-API-Key", "<api-key>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.prefetch.io/crawl/{id}")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["X-API-Key"] = '<api-key>'
response = http.request(request)
puts response.read_body{
"success": true,
"data": {
"id": "crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e",
"status": "running",
"url": "https://docs.stripe.com",
"total": 38,
"completed": 12,
"failed": 1,
"credits_used": 36,
"created_at": "2026-08-20T09:14:03.221Z",
"started_at": "2026-08-20T09:14:04.010Z",
"finished_at": null,
"expires_at": "2026-08-21T09:14:03.221Z",
"error": null,
"next": "/crawl/crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e?skip=25&limit=25",
"data": [
{
"url": "https://docs.stripe.com/payments",
"depth": 0,
"status_code": 200,
"ok": true,
"error": null,
"final_url": "https://docs.stripe.com/payments",
"domain": "docs.stripe.com",
"metadata": {
"title": "Payments",
"description": "Accept payments online.",
"language": "en"
},
"markdown": {
"content": "# Payments\n\nAccept payments online, in person, and around the world...",
"char_count": 2841,
"word_count": 402,
"truncated": false,
"main_content_only": true
}
}
]
},
"meta": {
"requestId": "d6a5f4e3-ac9d-4e6f-b04b-3c5d7e9f1a32",
"durationMs": 96
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}Overview
Returns where a crawl has got to, plus a page of results in completion order. Poll it untilstatus is completed.
Results are paginated. Each response carries a next link; follow it until next is null to read every page. Results stay readable for 24 hours after a crawl finishes, then expire.
Example request
curl "https://api.prefetch.io/crawl/crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e" \
-H "X-API-Key: $PREFETCH_API_KEY"
const id = "crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e";
const headers = { "X-API-Key": process.env.PREFETCH_API_KEY };
// Poll until the crawl finishes, then read every page of results.
let status;
do {
await new Promise((r) => setTimeout(r, 5000));
const res = await fetch(`https://api.prefetch.io/crawl/${id}`, { headers });
({ data: status } = await res.json());
console.log(`${status.completed}/${status.total} pages`);
} while (status.status === "pending" || status.status === "running");
const pages = [...status.data];
let next = status.next;
while (next) {
const res = await fetch(`https://api.prefetch.io${next}`, { headers });
const { data } = await res.json();
pages.push(...data.data);
next = data.next;
}
console.log(`${pages.length} pages collected`);
import os, time, requests
crawl_id = "crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e"
headers = {"X-API-Key": os.environ["PREFETCH_API_KEY"]}
base = "https://api.prefetch.io"
while True:
data = requests.get(f"{base}/crawl/{crawl_id}", headers=headers).json()["data"]
print(f"{data['completed']}/{data['total']} pages")
if data["status"] not in ("pending", "running"):
break
time.sleep(5)
pages, nxt = list(data["data"]), data["next"]
while nxt:
data = requests.get(base + nxt, headers=headers).json()["data"]
pages.extend(data["data"])
nxt = data["next"]
Example response
{
"success": true,
"data": {
"id": "crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e",
"status": "running",
"url": "https://docs.stripe.com",
"total": 38,
"completed": 12,
"failed": 1,
"credits_used": 36,
"created_at": "2026-08-20T09:14:03.221Z",
"started_at": "2026-08-20T09:14:04.010Z",
"finished_at": null,
"expires_at": "2026-08-21T09:14:03.221Z",
"error": null,
"next": "/crawl/crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e?skip=25&limit=25",
"data": [
{
"url": "https://docs.stripe.com/payments",
"depth": 0,
"status_code": 200,
"ok": true,
"error": null,
"final_url": "https://docs.stripe.com/payments",
"domain": "docs.stripe.com",
"metadata": {
"title": "Payments",
"description": "Accept payments online.",
"language": "en"
},
"markdown": {
"content": "# Payments\n\nAccept payments online, in person, and around the world...",
"char_count": 2841,
"word_count": 402,
"truncated": false,
"main_content_only": true
}
}
]
},
"meta": {
"requestId": "d6a5f4e3-ac9d-4e6f-b04b-3c5d7e9f1a32",
"durationMs": 96
}
}
Statuses
| Status | Meaning |
|---|---|
pending | Queued, not started. |
running | In progress. data already holds the pages finished so far. |
completed | Finished. |
failed | Stopped early — see error. Pages already stored are still readable. |
cancelled | You called DELETE /crawl/{id}. |
total is the frontier size as currently known, not a final count. A crawl discovers pages as it goes, so this number rises until the frontier is exhausted or limit is reached. Do not treat completed === total as “finished” — check status.Page objects
Each entry indata carries the crawl’s bookkeeping plus the same fields GET /scrape returns for that page:
| Field | Meaning |
|---|---|
url | The page that was crawled. |
depth | How many links from the start URL. Sitemap-seeded pages are 0. |
ok | false when the page could not be fetched. |
error | Why it failed, when ok is false. |
status_code, metadata, markdown, … | Exactly as in GET /scrape, shaped by scrape_options. |
Billing
This endpoint is where crawled pages are charged: each call bills for the pages that completed since your previous call, at 3 credits each. That makes polling safe — calling it ten times while a crawl runs costs exactly the same as calling it once at the end.credits_used on the response shows the running total for the crawl.
Related
POST /crawl
Start a crawl and configure its scope.
DELETE /crawl/{id}
Stop a running crawl. Free.
Authorizations
Your Prefetch API key. Obtain one from the dashboard.
Path Parameters
The crawl id returned by POST /crawl.
Example:
"crw_9f2c1a7b4e0d4c3a8b1e5f7d2c9a0b3e"
Query Parameters
Number of result pages to skip. Use the next link rather than building this by hand.
Required range:
x >= 0Number of crawled pages to return per request.
Required range:
1 <= x <= 100