curl --request POST \
--url https://api.scrapebento.com/v1/public/scrape/batch \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"urls": [
"https://example.com",
"https://example.org"
],
"format": "markdown"
}
'import requests
url = "https://api.scrapebento.com/v1/public/scrape/batch"
payload = {
"urls": ["https://example.com", "https://example.org"],
"format": "markdown"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({urls: ['https://example.com', 'https://example.org'], format: 'markdown'})
};
fetch('https://api.scrapebento.com/v1/public/scrape/batch', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.scrapebento.com/v1/public/scrape/batch",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'urls' => [
'https://example.com',
'https://example.org'
],
'format' => 'markdown'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.scrapebento.com/v1/public/scrape/batch"
payload := strings.NewReader("{\n \"urls\": [\n \"https://example.com\",\n \"https://example.org\"\n ],\n \"format\": \"markdown\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.scrapebento.com/v1/public/scrape/batch")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"urls\": [\n \"https://example.com\",\n \"https://example.org\"\n ],\n \"format\": \"markdown\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.scrapebento.com/v1/public/scrape/batch")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"urls\": [\n \"https://example.com\",\n \"https://example.org\"\n ],\n \"format\": \"markdown\"\n}"
response = http.request(request)
puts response.read_body{
"data": {
"jobs": [
{
"url": "https://example.com",
"jobId": "job_batch_1",
"status": "queued",
"pollUrl": "/v1/public/jobs/job_batch_1",
"creditsCharged": 1
},
{
"url": "https://example.org",
"error": {
"code": "invalid_url",
"message": "host is not a valid public target"
}
}
]
}
}Queue a batch of URL scrapes
Requires api.write. Accepts up to 50 URLs and always returns an async
batch response. Each item includes either a queued job record, an
inline validation/dispatch error, or — when a live cached result
exists and fresh is not set — an inline cached result
(status: succeeded, cached: true, result). A cached item is
recorded as a job and charged at the normal rate; it differs from a
queued one only in carrying its result instead of a pollUrl.
The per-organization in-flight cap applies to the batch as a whole:
if the URLs would take the organization’s queued + running jobs past
it, the whole batch is refused with 429 too_many_in_flight and
nothing is charged. Entries an idempotent retry would only replay are
not counted.
curl --request POST \
--url https://api.scrapebento.com/v1/public/scrape/batch \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"urls": [
"https://example.com",
"https://example.org"
],
"format": "markdown"
}
'import requests
url = "https://api.scrapebento.com/v1/public/scrape/batch"
payload = {
"urls": ["https://example.com", "https://example.org"],
"format": "markdown"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({urls: ['https://example.com', 'https://example.org'], format: 'markdown'})
};
fetch('https://api.scrapebento.com/v1/public/scrape/batch', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.scrapebento.com/v1/public/scrape/batch",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'urls' => [
'https://example.com',
'https://example.org'
],
'format' => 'markdown'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.scrapebento.com/v1/public/scrape/batch"
payload := strings.NewReader("{\n \"urls\": [\n \"https://example.com\",\n \"https://example.org\"\n ],\n \"format\": \"markdown\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.scrapebento.com/v1/public/scrape/batch")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"urls\": [\n \"https://example.com\",\n \"https://example.org\"\n ],\n \"format\": \"markdown\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.scrapebento.com/v1/public/scrape/batch")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"urls\": [\n \"https://example.com\",\n \"https://example.org\"\n ],\n \"format\": \"markdown\"\n}"
response = http.request(request)
puts response.read_body{
"data": {
"jobs": [
{
"url": "https://example.com",
"jobId": "job_batch_1",
"status": "queued",
"pollUrl": "/v1/public/jobs/job_batch_1",
"creditsCharged": 1
},
{
"url": "https://example.org",
"error": {
"code": "invalid_url",
"message": "host is not a valid public target"
}
}
]
}
}Authorizations
Send a workspace API token in the Authorization: Bearer <token> header.
Keys carry the satk_ prefix and spend your organization's credit
balance. GET /v1/public/balance reports what is left.
Test-mode keys (satk_test_), which drew on a separate self-refilling
allowance, have been retired and no longer authenticate.
Headers
Optional deduplication key, scoped to your organization. Re-sending a request with a key that was already used returns the original job instead of creating and charging for a second one.
Supersedes the idempotencyKey body field, which remains supported as
a deprecated alias. If both are present, this header wins.
255Body
Batch requests always return asynchronously, even if wait is supplied.
1 - 50 elementsWait for a terminal response before falling back to async job polling.
Optional client-requested sync wait timeout, bounded by the server maximum.
Optional public callback URL for terminal job delivery. Available to every organization.
Naming a URL here registers it as a webhook endpoint if it is not
registered already, so the signing secret exists before the first
delivery does. Fetch it from
GET /v1/public/webhooks/endpoints/{endpointId}/secret, or register
the endpoint up front with POST /v1/public/webhooks/endpoints to
have the secret in hand before you run anything.
Deliveries carry X-Scrapebento-Signature, an HMAC-SHA256 over
"<timestamp>.<body>" under that endpoint's own secret, so the
secret you need to verify your deliveries cannot forge anyone
else's. Every attempt is recorded in
GET /v1/public/webhooks/deliveries and can be re-sent from there.
Deprecated alias for the Idempotency-Key header, which is the
conventional transport and takes precedence when both are sent.
Still honoured so existing integrations keep working.
255How long this result may be reused, in seconds. Capped at 7 days (604800); larger values are clamped.
Omit it to accept the per-kind server default. Slow-moving kinds
cache by default — infra and brand for 6h, seo, sitemap,
site, and producthunt for 1h, and hn_search, hn_post, and
rss for 5m. Everything else defaults to no caching.
A cached result is charged at the same rate as a fresh scrape, so
the TTL is how stale an answer you can be billed for. Send 0 to
prevent this result being cached, and fresh: true to bypass an
existing cache entry on read.
0 <= x <= 604800Bypass cache and force a new scrape or search execution.
raw, markdown, json, metadata, screenshot, resolve Endpoint-specific options passed through to the worker/sidecar.
Response
Batch accepted.
Show child attributes
Show child attributes