Download Crawl Results as ZIP
curl --request GET \
--url https://api.spidra.io/api/crawl/{jobId}/download \
--header 'Authorization: Bearer <token>'import requests
url = "https://api.spidra.io/api/crawl/{jobId}/download"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://api.spidra.io/api/crawl/{jobId}/download', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.spidra.io/api/crawl/{jobId}/download",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.spidra.io/api/crawl/{jobId}/download"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.spidra.io/api/crawl/{jobId}/download")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.spidra.io/api/crawl/{jobId}/download")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body"<string>"{
"status": "error",
"message": "Access token invalid or expired"
}{
"status": "error",
"message": "Unauthorized access or job not found"
}{
"status": "error",
"message": "No successful pages found for this crawl"
}Crawl Endpoints
Download Crawl Results
Download the results of a completed crawl job as a ZIP archive containing HTML, markdown, and extracted data files.
GET
/
crawl
/
{jobId}
/
download
Download Crawl Results as ZIP
curl --request GET \
--url https://api.spidra.io/api/crawl/{jobId}/download \
--header 'Authorization: Bearer <token>'import requests
url = "https://api.spidra.io/api/crawl/{jobId}/download"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://api.spidra.io/api/crawl/{jobId}/download', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.spidra.io/api/crawl/{jobId}/download",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.spidra.io/api/crawl/{jobId}/download"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.spidra.io/api/crawl/{jobId}/download")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.spidra.io/api/crawl/{jobId}/download")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body"<string>"{
"status": "error",
"message": "Access token invalid or expired"
}{
"status": "error",
"message": "Unauthorized access or job not found"
}{
"status": "error",
"message": "No successful pages found for this crawl"
}Downloads a ZIP archive of all successfully crawled pages from a completed job. Each page is saved as one or more files inside the archive, organized by hostname and path. Use the
If you omit the
When multiple content types are requested, each page gets its own folder:
include parameter to control which content types are bundled in the ZIP.
Content Types
| Value | What is included |
|---|---|
html | Raw HTML file for each page |
markdown | Markdown version of each page |
data | AI-extracted data in JSON, CSV, or Markdown format (format is auto-detected from your transformInstruction) |
include parameter, all three types are included by default.
Example Requests
# Download everything (HTML, markdown, and extracted data)
curl -OJ "https://api.spidra.io/api/crawl/abc-123/download" \
-H "Authorization: Bearer YOUR_API_KEY"
# Download only the extracted data
curl -OJ "https://api.spidra.io/api/crawl/abc-123/download?include=data" \
-H "Authorization: Bearer YOUR_API_KEY"
# Download markdown and data only
curl -OJ "https://api.spidra.io/api/crawl/abc-123/download?include=markdown,data" \
-H "Authorization: Bearer YOUR_API_KEY"
import requests
# Download everything
response = requests.get(
"https://api.spidra.io/api/crawl/abc-123/download",
headers={"Authorization": "Bearer YOUR_API_KEY"}
)
with open("crawl-abc-123.zip", "wb") as f:
f.write(response.content)
# Download only extracted data
response = requests.get(
"https://api.spidra.io/api/crawl/abc-123/download",
headers={"Authorization": "Bearer YOUR_API_KEY"},
params={"include": "data"}
)
with open("crawl-abc-123.zip", "wb") as f:
f.write(response.content)
import { writeFileSync } from "fs";
// Download everything
const response = await fetch(
"https://api.spidra.io/api/crawl/abc-123/download",
{ headers: { Authorization: "Bearer YOUR_API_KEY" } }
);
const buffer = await response.arrayBuffer();
writeFileSync("crawl-abc-123.zip", Buffer.from(buffer));
// Download only extracted data
const dataOnly = await fetch(
"https://api.spidra.io/api/crawl/abc-123/download?include=data",
{ headers: { Authorization: "Bearer YOUR_API_KEY" } }
);
const buf = await dataOnly.arrayBuffer();
writeFileSync("crawl-abc-123.zip", Buffer.from(buf));
ZIP Archive Structure
When a single content type is requested, files are placed at the root of the archive with appropriate extensions:crawl-abc-123.zip
example.com_blog_post-one.json
example.com_blog_post-two.json
crawl-abc-123.zip
example.com_blog_post-one/
data.json
index.html
markdown.md
example.com_blog_post-two/
data.json
index.html
markdown.md
Response
The response is a binary ZIP file with the following headers:| Header | Value |
|---|---|
Content-Type | application/zip |
Content-Disposition | attachment; filename=crawl-{jobId}.zip |
Only pages with
status: "success" are included in the download. If no successful pages exist, the API returns a 404 error.Authorizations
BearerAuthApiKeyAuth
Bearer authentication header of the form Bearer <token>, where <token> is your auth token.
Path Parameters
The ID of the completed crawl job to download
Query Parameters
Comma-separated list of content types to include in the ZIP. Accepted values: html, markdown, data. Defaults to all three. Example: include=data,markdown
Response
ZIP archive containing the crawl results
The response is of type file.
Was this page helpful?

