Get Crawled Pages
curl --request GET \
--url https://api.spidra.io/api/crawl/{jobId}/pages \
--header 'Authorization: Bearer <token>'import requests
url = "https://api.spidra.io/api/crawl/{jobId}/pages"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://api.spidra.io/api/crawl/{jobId}/pages', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.spidra.io/api/crawl/{jobId}/pages",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.spidra.io/api/crawl/{jobId}/pages"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.spidra.io/api/crawl/{jobId}/pages")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.spidra.io/api/crawl/{jobId}/pages")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body{
"pages": [
{
"id": "page-uuid-1",
"url": "https://example.com/blog/post-1",
"title": "First Post",
"status": "success",
"data": {
"title": "First Post",
"author": "John",
"date": "2025-01-01"
},
"error_message": null,
"html": "https://storage.spidra.io/signed/crawl/abc-123/page1.html?...",
"markdown": "https://storage.spidra.io/signed/crawl/abc-123/page1.md?...",
"created_at": "2025-12-17T15:00:00Z"
},
{
"id": "page-uuid-2",
"url": "https://example.com/blog/broken-page",
"title": null,
"status": "failed",
"data": null,
"error_message": "AI transformation failed: content too short to extract",
"html": null,
"markdown": null,
"created_at": "2025-12-17T15:01:30Z"
}
]
}{
"status": "error",
"message": "Access token invalid or expired"
}{
"status": "error",
"message": "Unauthorized access or job not found"
}Crawl Endpoints
Get Crawled Pages
Retrieve all pages from a completed crawl job, including extracted data and signed URLs to the raw HTML and markdown files.
GET
/
crawl
/
{jobId}
/
pages
Get Crawled Pages
curl --request GET \
--url https://api.spidra.io/api/crawl/{jobId}/pages \
--header 'Authorization: Bearer <token>'import requests
url = "https://api.spidra.io/api/crawl/{jobId}/pages"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://api.spidra.io/api/crawl/{jobId}/pages', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.spidra.io/api/crawl/{jobId}/pages",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.spidra.io/api/crawl/{jobId}/pages"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.spidra.io/api/crawl/{jobId}/pages")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.spidra.io/api/crawl/{jobId}/pages")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body{
"pages": [
{
"id": "page-uuid-1",
"url": "https://example.com/blog/post-1",
"title": "First Post",
"status": "success",
"data": {
"title": "First Post",
"author": "John",
"date": "2025-01-01"
},
"error_message": null,
"html": "https://storage.spidra.io/signed/crawl/abc-123/page1.html?...",
"markdown": "https://storage.spidra.io/signed/crawl/abc-123/page1.md?...",
"created_at": "2025-12-17T15:00:00Z"
},
{
"id": "page-uuid-2",
"url": "https://example.com/blog/broken-page",
"title": null,
"status": "failed",
"data": null,
"error_message": "AI transformation failed: content too short to extract",
"html": null,
"markdown": null,
"created_at": "2025-12-17T15:01:30Z"
}
]
}{
"status": "error",
"message": "Access token invalid or expired"
}{
"status": "error",
"message": "Unauthorized access or job not found"
}Returns every page processed by a crawl job. Call this once the job status is
The
What
completed.
Each page record includes the extracted content in data, plus signed URLs to the original HTML snapshot and markdown version stored by Spidra.
Example Request
curl https://api.spidra.io/api/crawl/abc-123/pages \
-H "Authorization: Bearer YOUR_API_KEY"
import requests
response = requests.get(
"https://api.spidra.io/api/crawl/abc-123/pages",
headers={"Authorization": "Bearer YOUR_API_KEY"}
)
const response = await fetch("https://api.spidra.io/api/crawl/abc-123/pages", {
headers: { Authorization: "Bearer YOUR_API_KEY" }
});
Response Fields
| Field | Type | Description |
|---|---|---|
pages | array | All pages processed by this job, including failed ones. |
pages[].id | string | Unique page ID. Pass this to POST /crawl//extract to re-run extraction on a specific page. |
pages[].url | string | The URL of this page. |
pages[].title | string | Page title as detected during crawling. |
pages[].status | string | success or failed. |
pages[].data | any | Extracted content for this page. When a transformInstruction or schema was provided, this contains AI-extracted structured data. When neither was set, this is the raw page markdown — no AI was used. |
pages[].error_message | string or null | Error details when status is failed. |
pages[].html | string or null | Signed URL to the raw HTML snapshot. Valid for 1 hour. |
pages[].markdown | string or null | Signed URL to the markdown version of this page. Valid for 1 hour. |
pages[].created_at | string | ISO 8601 timestamp when this page was processed. |
Example Response
{
"pages": [
{
"id": "page-uuid-1",
"url": "https://example.com/blog/how-to-scrape",
"title": "How to Scrape the Web Without Getting Blocked",
"status": "success",
"data": {
"title": "How to Scrape the Web Without Getting Blocked",
"author": "Jane Smith",
"published": "2025-11-20",
"summary": "A guide to rotating proxies and handling JavaScript-heavy pages."
},
"error_message": null,
"html": "https://storage.spidra.io/signed/...",
"markdown": "https://storage.spidra.io/signed/...",
"created_at": "2025-12-17T15:02:10Z"
},
{
"id": "page-uuid-2",
"url": "https://example.com/blog/javascript-rendering",
"title": "JavaScript Rendering Explained",
"status": "failed",
"data": null,
"error_message": "AI transformation failed: content too short to extract",
"html": null,
"markdown": null,
"created_at": "2025-12-17T15:02:55Z"
}
]
}
The data Field
What data contains depends on how you configured the job:
- With
transformInstruction—datais whatever the AI extracted based on your prompt. It could be a string, an object, or structured JSON depending on what you asked for. - With
schema—datais a JSON object matching the schema you defined, with all fields present. - With neither —
datais the raw page markdown. No AI was involved and no token credits were charged. ThehtmlandmarkdownURLs point to the same content in its original format.
Handling Failed Pages
Pages withstatus: "failed" still appear in the response. The error_message explains what went wrong. To re-run extraction on a failed page without re-crawling the site, use the extract endpoint.
The
html and markdown URLs expire after one hour. If you need permanent access to the raw files, use Download Crawl Results to get a full ZIP archive.Was this page helpful?

