curl --request GET \
--url https://api.prefetch.io/scrape \
--header 'X-API-Key: <api-key>'import requests
url = "https://api.prefetch.io/scrape"
headers = {"X-API-Key": "<api-key>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {'X-API-Key': '<api-key>'}};
fetch('https://api.prefetch.io/scrape', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.prefetch.io/scrape",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"X-API-Key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.prefetch.io/scrape"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("X-API-Key", "<api-key>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.prefetch.io/scrape")
.header("X-API-Key", "<api-key>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.prefetch.io/scrape")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["X-API-Key"] = '<api-key>'
response = http.request(request)
puts response.read_body{
"success": true,
"data": {
"url": "https://stripe.com",
"final_url": "https://stripe.com",
"domain": "stripe.com",
"status_code": 200,
"metadata": {
"source_url": "https://stripe.com",
"final_url": "https://stripe.com",
"status_code": 200,
"title": "Stripe | Financial Infrastructure to Grow Your Revenue",
"description": "Stripe powers online payment processing for internet businesses.",
"language": "en",
"canonical_url": "https://stripe.com",
"favicon": "https://stripe.com/favicon.ico",
"author": null,
"site_name": "Stripe",
"published_time": null,
"modified_time": null,
"robots": null,
"keywords": [],
"open_graph": {
"title": "Stripe",
"type": "website"
},
"twitter": {
"card": "summary_large_image"
},
"json_ld": [],
"headings": [
{
"level": 1,
"text": "Financial infrastructure to grow your revenue"
}
]
},
"markdown": {
"content": "# Financial infrastructure to grow your revenue\n\nJoin the millions of companies that use Stripe to accept payments online...",
"char_count": 4821,
"word_count": 702,
"truncated": false,
"main_content_only": true
},
"links": {
"internal": [
{
"url": "https://stripe.com/pricing",
"text": "Pricing",
"title": null,
"rel": null
}
],
"external": [],
"count": 84
}
},
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 2140
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}GET /scrape
Turn any page into clean markdown, an LLM summary, structured JSON, links, or images. Costs 3 credits.
curl --request GET \
--url https://api.prefetch.io/scrape \
--header 'X-API-Key: <api-key>'import requests
url = "https://api.prefetch.io/scrape"
headers = {"X-API-Key": "<api-key>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {'X-API-Key': '<api-key>'}};
fetch('https://api.prefetch.io/scrape', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.prefetch.io/scrape",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"X-API-Key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.prefetch.io/scrape"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("X-API-Key", "<api-key>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.prefetch.io/scrape")
.header("X-API-Key", "<api-key>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.prefetch.io/scrape")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["X-API-Key"] = '<api-key>'
response = http.request(request)
puts response.read_body{
"success": true,
"data": {
"url": "https://stripe.com",
"final_url": "https://stripe.com",
"domain": "stripe.com",
"status_code": 200,
"metadata": {
"source_url": "https://stripe.com",
"final_url": "https://stripe.com",
"status_code": 200,
"title": "Stripe | Financial Infrastructure to Grow Your Revenue",
"description": "Stripe powers online payment processing for internet businesses.",
"language": "en",
"canonical_url": "https://stripe.com",
"favicon": "https://stripe.com/favicon.ico",
"author": null,
"site_name": "Stripe",
"published_time": null,
"modified_time": null,
"robots": null,
"keywords": [],
"open_graph": {
"title": "Stripe",
"type": "website"
},
"twitter": {
"card": "summary_large_image"
},
"json_ld": [],
"headings": [
{
"level": 1,
"text": "Financial infrastructure to grow your revenue"
}
]
},
"markdown": {
"content": "# Financial infrastructure to grow your revenue\n\nJoin the millions of companies that use Stripe to accept payments online...",
"char_count": 4821,
"word_count": 702,
"truncated": false,
"main_content_only": true
},
"links": {
"internal": [
{
"url": "https://stripe.com/pricing",
"text": "Pricing",
"title": null,
"rel": null
}
],
"external": [],
"count": 84
}
},
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 2140
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}Overview
/scrape fetches one URL and gives it back to you in whichever formats you ask for. It handles the parts that make scraping tedious:
- JavaScript rendering — pages that need a browser get one, automatically. Static pages take the fast path.
- Bot walls — blocked requests escalate through stealth and proxy tiers before giving up.
- Boilerplate — navigation, sidebars, footers, and cookie banners are stripped by default.
- Absolute URLs — every link and image is resolved against the page, so the output still works once you store it somewhere else.
formats as a comma-separated list. Each format you request appears as its own object on data. Formats you do not request are absent, not null — so if (data.markdown) is a reliable check.
| Format | You get |
|---|---|
markdown | GitHub Flavored Markdown, plus character and word counts. The default. |
summary | A natural-language summary of the page. |
html | Cleaned HTML: scripts, styles, and inline event handlers removed. |
raw_html | The source exactly as fetched. |
links | Every link, split into internal and external. |
images | Every image, including srcset and lazy-loaded sources. |
json | Structured data matching your own prompt or JSON Schema. |
metadata is always present, whatever you request.
Example requests
curl "https://api.prefetch.io/scrape?url=https://stripe.com" \
-H "X-API-Key: $PREFETCH_API_KEY"
curl "https://api.prefetch.io/scrape?url=https://stripe.com&formats=markdown,links" \
-H "X-API-Key: $PREFETCH_API_KEY"
curl "https://api.prefetch.io/scrape?url=https://stripe.com&only_main_content=false" \
-H "X-API-Key: $PREFETCH_API_KEY"
const params = new URLSearchParams({
url: "https://stripe.com",
formats: "markdown,links",
});
const res = await fetch(`https://api.prefetch.io/scrape?${params}`, {
headers: { "X-API-Key": process.env.PREFETCH_API_KEY },
});
const { data } = await res.json();
console.log(data.markdown.content);
console.log(`${data.links.count} links found`);
import requests, os
r = requests.get(
"https://api.prefetch.io/scrape",
params={"url": "https://stripe.com", "formats": "markdown,links"},
headers={"X-API-Key": os.environ["PREFETCH_API_KEY"]},
)
data = r.json()["data"]
print(data["markdown"]["content"])
Example response
{
"success": true,
"data": {
"url": "https://stripe.com",
"final_url": "https://stripe.com",
"domain": "stripe.com",
"status_code": 200,
"metadata": {
"source_url": "https://stripe.com",
"final_url": "https://stripe.com",
"status_code": 200,
"title": "Stripe | Financial Infrastructure to Grow Your Revenue",
"description": "Stripe powers online payment processing for internet businesses.",
"language": "en",
"canonical_url": "https://stripe.com",
"favicon": "https://stripe.com/favicon.ico",
"site_name": "Stripe",
"keywords": [],
"open_graph": { "title": "Stripe", "type": "website" },
"twitter": { "card": "summary_large_image" },
"json_ld": [],
"headings": [
{ "level": 1, "text": "Financial infrastructure to grow your revenue" }
]
},
"markdown": {
"content": "# Financial infrastructure to grow your revenue\n\nJoin the millions of companies that use Stripe...",
"char_count": 4821,
"word_count": 702,
"truncated": false,
"main_content_only": true
},
"links": {
"internal": [
{ "url": "https://stripe.com/pricing", "text": "Pricing", "title": null, "rel": null }
],
"external": [],
"count": 84
}
},
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 2140
}
}
Main content isolation
only_main_content defaults to true. It removes navigation, headers, footers, sidebars, share widgets, and cookie bars before rendering markdown, html, summary, and json.
When a page is not article-shaped — a product grid, a landing page — isolation would have to guess which fragment is “the content”. Rather than guess, it returns the full page body, sets main_content_only: false, and tells you in warnings:
{
"markdown": { "main_content_only": false, "...": "..." },
"warnings": ["Main content could not be isolated; returned the full page body instead."]
}
links and images are always read from the full document, even when only_main_content is true. If you are mapping a site, the navigation links are exactly the ones you want.include_selectors to name the content, or exclude_selectors to name what to drop:
curl "https://api.prefetch.io/scrape?url=https://example.com/post&include_selectors=article,.post-body" \
-H "X-API-Key: $PREFETCH_API_KEY"
Structured extraction with json
Ask for the json format with a plain-language json_prompt, a json_schema, or both. Requesting json without either returns a 400.
curl -G "https://api.prefetch.io/scrape" \
--data-urlencode "url=https://example.com/product" \
--data-urlencode "formats=json" \
--data-urlencode "json_prompt=product name, price and availability" \
-H "X-API-Key: $PREFETCH_API_KEY"
curl -G "https://api.prefetch.io/scrape" \
--data-urlencode "url=https://example.com/product" \
--data-urlencode "formats=json" \
--data-urlencode 'json_schema={"type":"object","properties":{"name":{"type":"string"},"price":{"type":"number"}}}' \
-H "X-API-Key: $PREFETCH_API_KEY"
{
"json": {
"data": {
"product_name": "Widget Pro",
"price": "$49.00",
"availability": "In stock"
},
"model": "gpt-4.1-mini"
}
}
null rather than being filled in from the model’s own knowledge.
Partial failures
The LLM-backed formats (summary and json) can fail on their own without losing the rest of the scrape. When that happens the format returns null values and the reason appears in warnings:
{
"markdown": { "content": "# Still here...", "...": "..." },
"summary": { "text": null, "model": null },
"warnings": ["Summary failed: Request timed out"]
}
success: true, and you are still charged — the page was fetched.
Content size
markdown, html, and raw_html are each capped at 1 MB. Content that hits the cap is cut on a character boundary and flagged:
{ "raw_html": { "char_count": 1000000, "truncated": true, "...": "..." } }
Related
GET /map
POST /crawl
Authorizations
Your Prefetch API key. Obtain one from the dashboard.
Query Parameters
The website URL to process. https:// is prepended automatically if no protocol is provided.
"https://stripe.com"
Comma-separated list of output formats. Each requested format appears as its own object on data; formats you do not request are absent, not null.
"markdown,links"
Strip navigation, sidebars, footers, and cookie banners before rendering markdown, html, summary, and json. links and images always come from the full document.
Keep hyperlinks in the markdown and html output. When false, link text is kept and the target is dropped.
Keep images in the markdown and html output.
Keep inline data: images. Off by default because base64 payloads are large and carry no meaning as text.
Comma-separated CSS selectors to keep. Overrides main-content detection entirely.
"article,.post-body"
Comma-separated CSS selectors to remove before rendering.
".cookie-bar,#promo"
What to extract, in plain language. Required for the json format unless json_schema is given.
"product name, price and availability"
A JSON Schema, passed as a JSON string, describing the object the json format should return.
Word budget for the summary format.
20 <= x <= 400