NewIntroducing semantic snapshotsPair every capture with structured DOM data →

For teams doing this to a whole site

Work through a site without writing a job runner

Point a durable crawl at a seed URL, or hand a batch a list, and collect every page's content and a picture of it.

The problem

One page is a request. Nine hundred pages is infrastructure: a queue, retries that do not double-charge, somewhere to put partial results, a way to answer "is it done yet", and a plan for the run that dies at page 600 because a worker timed out. Most of that is not the problem you set out to solve.

How domscout handles it

Point a crawl at a seed URL with the origins and path patterns it may follow, or hand a batch an explicit list. The job is durable: it survives a worker dying, each child reports its own result, quota is charged when a child starts rather than when the parent is created, and an idempotency key means a retried submission is the same run and not a second one.

Where it shows up

  • First-time ingestion of a documentation site or knowledge base.
  • Periodic refresh of a catalogue you already index.
  • Bulk migration work where you need every page's content and a picture of it.

The request

POST /crawl. Crawl and batch start at Business. A crawl stops at 500 pages and depth 5, needs an explicit HTTPS origin allowlist, and always obeys robots.txt.

cURL
curl -X POST https://api.domscout.io/crawl \
  -H "x-api-key: YOUR_API_KEY" \
  -H "Content-Type: application/json" \
  -d '{
    "seedUrl": "https://docs.example.com",
    "allowedOrigins": ["https://docs.example.com"],
    "maxPages": 200,
    "capture": {"extractMarkdown": true}
  }'
Node.js
const response = await fetch("https://api.domscout.io/crawl", {
  method: "POST",
  headers: { "Content-Type": "application/json", "x-api-key": "YOUR_API_KEY" },
  body: JSON.stringify({
    seedUrl: "https://docs.example.com",
    allowedOrigins: ["https://docs.example.com"],
    maxPages: 200,
    capture: { extractMarkdown: true }
  }),
});
console.log(await response.json());
Python
import requests

response = requests.post(
    "https://api.domscout.io/crawl",
    headers={"x-api-key": "YOUR_API_KEY"},
    json={
        "seedUrl": "https://docs.example.com",
        "allowedOrigins": ["https://docs.example.com"],
        "maxPages": 200,
        "capture": {"extractMarkdown": True}
    },
)
print(response.json())
Go
package main

import (
	"bytes"
	"encoding/json"
	"fmt"
	"io"
	"net/http"
)

func main() {
	body, _ := json.Marshal(map[string]any{
		"seedUrl": "https://docs.example.com",
		"allowedOrigins": []any{"https://docs.example.com"},
		"maxPages": 200,
		"capture": map[string]any{"extractMarkdown": true},
	})
	req, _ := http.NewRequest("POST", "https://api.domscout.io/crawl", bytes.NewReader(body))
	req.Header.Set("Content-Type", "application/json")
	req.Header.Set("x-api-key", "YOUR_API_KEY")

	res, err := http.DefaultClient.Do(req)
	if err != nil {
		panic(err)
	}
	defer res.Body.Close()

	raw, _ := io.ReadAll(res.Body)
	fmt.Println(string(raw))
}
PHP
<?php
$ch = curl_init('https://api.domscout.io/crawl');
curl_setopt_array($ch, [
    CURLOPT_POST => true,
    CURLOPT_RETURNTRANSFER => true,
    CURLOPT_HTTPHEADER => ['Content-Type: application/json', 'x-api-key: YOUR_API_KEY'],
    CURLOPT_POSTFIELDS => json_encode([
        'seedUrl' => 'https://docs.example.com',
        'allowedOrigins' => ['https://docs.example.com'],
        'maxPages' => 200,
        'capture' => ['extractMarkdown' => true]
    ]),
]);
$response = curl_exec($ch);
if ($response === false) {
    throw new RuntimeException(curl_error($ch));
}
echo $response;
Ruby
require "json"
require "net/http"

response = Net::HTTP.post(
  URI("https://api.domscout.io/crawl"),
  {
    seedUrl: "https://docs.example.com",
    allowedOrigins: ["https://docs.example.com"],
    maxPages: 200,
    capture: { extractMarkdown: true }
  }.to_json,
  "Content-Type" => "application/json",
  "x-api-key" => "YOUR_API_KEY"
)
puts response.body
Java
import java.net.URI;
import java.net.http.HttpClient;
import java.net.http.HttpRequest;
import java.net.http.HttpResponse;

public class Capture {
    public static void main(String[] args) throws Exception {
        String body = """
            {
                "seedUrl": "https://docs.example.com",
                "allowedOrigins": ["https://docs.example.com"],
                "maxPages": 200,
                "capture": {"extractMarkdown": true}
            }
            """;
        HttpRequest request = HttpRequest.newBuilder(URI.create("https://api.domscout.io/crawl"))
            .header("Content-Type", "application/json")
            .header("x-api-key", "YOUR_API_KEY")
            .POST(HttpRequest.BodyPublishers.ofString(body))
            .build();
        HttpResponse<String> response = HttpClient.newHttpClient().send(request, HttpResponse.BodyHandlers.ofString());
        System.out.println(response.body());
    }
}
C#
using System.Net.Http.Json;

using var client = new HttpClient();
client.DefaultRequestHeaders.Add("x-api-key", "YOUR_API_KEY");

var response = await client.PostAsJsonAsync("https://api.domscout.io/crawl", new {
    seedUrl = "https://docs.example.com",
    allowedOrigins = new[] { "https://docs.example.com" },
    maxPages = 200,
    capture = new { extractMarkdown = true }
});
Console.WriteLine(await response.Content.ReadAsStringAsync());

Go further

Other use cases: Web access for AI agents · Website change monitoring · Web archiving and evidence · SEO regression checks · Thumbnails and link previews · PDFs from your own HTML · Structured data from web pages · Link preview checks · Responsive layout checks

Give your product a browser.

Get clean web content and visual proof into your workflow in minutes.