Map a site's URLs
curl --request GET \
--url https://api.prefetch.io/map \
--header 'X-API-Key: <api-key>'import requests
url = "https://api.prefetch.io/map"
headers = {"X-API-Key": "<api-key>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {'X-API-Key': '<api-key>'}};
fetch('https://api.prefetch.io/map', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.prefetch.io/map",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"X-API-Key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.prefetch.io/map"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("X-API-Key", "<api-key>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.prefetch.io/map")
.header("X-API-Key", "<api-key>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.prefetch.io/map")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["X-API-Key"] = '<api-key>'
response = http.request(request)
puts response.read_body{
"success": true,
"data": {
"url": "https://stripe.com",
"domain": "stripe.com",
"links": [
{
"url": "https://stripe.com/pricing",
"title": "Pricing",
"description": null,
"lastmod": "2026-07-02",
"source": "sitemap"
},
{
"url": "https://stripe.com/enterprise/pricing",
"title": null,
"description": null,
"lastmod": null,
"source": "sitemap"
}
],
"count": 2,
"total_discovered": 2,
"sources": {
"sitemap": 2,
"links": 0
},
"sitemaps_used": [
"https://stripe.com/sitemap/sitemap.xml"
]
},
"meta": {
"requestId": "b4e3d2c1-8a7b-4c3d-9e2f-1a3b5c7d9e1f",
"durationMs": 1830
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}Web extraction
GET /map
Discover every URL a site exposes, from its sitemaps and on-page links, without fetching them. Costs 3 credits.
GET
/
map
Map a site's URLs
curl --request GET \
--url https://api.prefetch.io/map \
--header 'X-API-Key: <api-key>'import requests
url = "https://api.prefetch.io/map"
headers = {"X-API-Key": "<api-key>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {'X-API-Key': '<api-key>'}};
fetch('https://api.prefetch.io/map', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.prefetch.io/map",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"X-API-Key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.prefetch.io/map"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("X-API-Key", "<api-key>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.prefetch.io/map")
.header("X-API-Key", "<api-key>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.prefetch.io/map")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["X-API-Key"] = '<api-key>'
response = http.request(request)
puts response.read_body{
"success": true,
"data": {
"url": "https://stripe.com",
"domain": "stripe.com",
"links": [
{
"url": "https://stripe.com/pricing",
"title": "Pricing",
"description": null,
"lastmod": "2026-07-02",
"source": "sitemap"
},
{
"url": "https://stripe.com/enterprise/pricing",
"title": null,
"description": null,
"lastmod": null,
"source": "sitemap"
}
],
"count": 2,
"total_discovered": 2,
"sources": {
"sitemap": 2,
"links": 0
},
"sitemaps_used": [
"https://stripe.com/sitemap/sitemap.xml"
]
},
"meta": {
"requestId": "b4e3d2c1-8a7b-4c3d-9e2f-1a3b5c7d9e1f",
"durationMs": 1830
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}{
"success": false,
"error": "Credit limit exceeded",
"meta": {
"requestId": "a3f2c1d4-7b6e-4f2a-9c1d-8e3f2a1b4c5d",
"durationMs": 1842
}
}Overview
/map answers “what pages does this site have?” without downloading any of them. It merges two sources:
- Sitemaps — found through
robots.txtfirst, then the conventional locations (/sitemap.xml,/sitemap_index.xml). Sitemap indexes are followed one level deep. Authoritative and cheap: one request can describe a whole site. - On-page links — every link on the URL you pass. The fallback for sites that publish no sitemap, and the only source that reflects what is actually linked.
lastmod and is the site’s own statement about its content, but the anchor text from the page is kept as a title.
Use it to plan a crawl, to find one page without guessing its path, or to check what a site has published recently.
Example requests
curl "https://api.prefetch.io/map?url=https://stripe.com" \
-H "X-API-Key: $PREFETCH_API_KEY"
curl "https://api.prefetch.io/map?url=https://stripe.com&search=pricing&limit=20" \
-H "X-API-Key: $PREFETCH_API_KEY"
curl "https://api.prefetch.io/map?url=https://stripe.com&sitemap=only" \
-H "X-API-Key: $PREFETCH_API_KEY"
const params = new URLSearchParams({
url: "https://stripe.com",
search: "pricing",
limit: "20",
});
const res = await fetch(`https://api.prefetch.io/map?${params}`, {
headers: { "X-API-Key": process.env.PREFETCH_API_KEY },
});
const { data } = await res.json();
console.log(`${data.count} of ${data.total_discovered} links`);
data.links.forEach((l) => console.log(l.url));
import requests, os
r = requests.get(
"https://api.prefetch.io/map",
params={"url": "https://stripe.com", "search": "pricing", "limit": 20},
headers={"X-API-Key": os.environ["PREFETCH_API_KEY"]},
)
for link in r.json()["data"]["links"]:
print(link["url"], link["source"])
Example response
{
"success": true,
"data": {
"url": "https://stripe.com",
"domain": "stripe.com",
"links": [
{
"url": "https://stripe.com/pricing",
"title": "Pricing",
"description": null,
"lastmod": "2026-07-02",
"source": "sitemap"
},
{
"url": "https://stripe.com/enterprise/pricing",
"title": null,
"description": null,
"lastmod": null,
"source": "sitemap"
}
],
"count": 2,
"total_discovered": 2,
"sources": { "sitemap": 2, "links": 0 },
"sitemaps_used": ["https://stripe.com/sitemap/sitemap.xml"]
},
"meta": {
"requestId": "b4e3d2c1-8a7b-4c3d-9e2f-1a3b5c7d9e1f",
"durationMs": 1830
}
}
Reading the response
| Field | Meaning |
|---|---|
count | Links returned, after limit was applied. |
total_discovered | Links found before limit. A much larger number means you are seeing a slice. |
sources | How many returned links came from each source. {"sitemap": 0} means the site publishes no usable sitemap. |
sitemaps_used | The sitemap documents that actually produced entries. |
source on each link | sitemap or links. |
lastmod on each link | Present only for sitemap entries whose sitemap declared one. |
Ordering and search
Withoutsearch, links come back shortest path first — so the homepage and top-level sections lead, and the deep long tail follows.
With search, results are filtered as well as ranked: links matching nothing are dropped, not just pushed down. A match in the URL path outranks one in the link text.
# Returns only pricing-related URLs
curl "https://api.prefetch.io/map?url=https://stripe.com&search=pricing" \
-H "X-API-Key: $PREFETCH_API_KEY"
Choosing sources
sitemap | Behavior |
|---|---|
include | Default. Fetches the page and the sitemaps, then merges. |
only | Sitemaps only. Skips the page fetch, so it is faster and works even when the homepage blocks bots. |
skip | On-page links only. Use when a site’s sitemap is stale and you want what is actually linked today. |
robots.txt is read for its Sitemap: entries regardless. /map fetches at most one page of the site, so nothing here is affected by Disallow rules.Scope
By default the result stays on the site you asked about, subdomains included, and URLs that differ only by query string collapse into one.include_subdomains=false— restrict to the exact host.blog.stripe.comis excluded when you mapstripe.com.ignore_query_params=false— keep?page=2and?page=3as separate entries.
Related
GET /scrape
Fetch the content of any URL you found. 3 credits.
POST /crawl
Fetch all of them in one job. 3 credits + 3 per page.
Authorizations
Your Prefetch API key. Obtain one from the dashboard.
Query Parameters
The website URL to process. https:// is prepended automatically if no protocol is provided.
Example:
"https://stripe.com"
Maximum number of links to return.
Required range:
1 <= x <= 5000Rank and filter results by relevance to this query. Links that match nothing are dropped.
Example:
"pricing"
Which sources to use. include merges sitemap entries with on-page links, skip uses on-page links only, only uses sitemaps only.
Available options:
include, skip, only Include links on subdomains of the target.
Treat URLs that differ only by query string as one page.