NewIntroducing semantic snapshotsPair every capture with structured DOM data →

For teams feeding a database

Get the fields you need from a page, not the whole page

Name each field and its CSS selector, and get JSON back from the rendered page, with every field marked found or missing.

The problem

A price, a stock status, a spec table, a list of links. When you need a handful of values from a page, a full scrape leaves you writing a parser, and a parser that fails quietly returns an empty string that looks like a real answer.

How domscout handles it

Send extract.fields: a name for each field, a CSS selector, and a type (text, number, boolean, attribute, html, url or list). The page is rendered first, so values that arrive with JavaScript are there to be read, and a number field drops currency symbols and thousands separators before it parses. Each field comes back with a status of found, missing or invalid_selector, so an absent value cannot pass for an empty one, and strict: true fails the whole request when a field marked required is missing.

Where it shows up

  • Price and stock checks on product pages you track.
  • Filling records from pages you would otherwise copy by hand.
  • Collecting the same fields across a site with a crawl or a batch, on Business and above.

The request

POST /scrape. extract starts at Pro and adds 1 credit to the capture, so each page costs 2. The fields come back under analysis.extraction.fields.

cURL
curl -X POST https://api.domscout.io/scrape \
  -H "x-api-key: YOUR_API_KEY" \
  -H "Content-Type: application/json" \
  -d '{
    "url": "https://example.com/product/42",
    "extract": {
      "fields": {
        "name": {"selector": "h1", "type": "text"},
        "price": {"selector": ".price", "type": "number"},
        "image": {
          "selector": "img.product",
          "type": "attribute",
          "attribute": "src"
        }
      }
    }
  }'
Node.js
const response = await fetch("https://api.domscout.io/scrape", {
  method: "POST",
  headers: { "Content-Type": "application/json", "x-api-key": "YOUR_API_KEY" },
  body: JSON.stringify({
    url: "https://example.com/product/42",
    extract: {
      fields: {
        name: { selector: "h1", type: "text" },
        price: { selector: ".price", type: "number" },
        image: { selector: "img.product", type: "attribute", attribute: "src" }
      }
    }
  }),
});
console.log(await response.json());
Python
import requests

response = requests.post(
    "https://api.domscout.io/scrape",
    headers={"x-api-key": "YOUR_API_KEY"},
    json={
        "url": "https://example.com/product/42",
        "extract": {
            "fields": {
                "name": {"selector": "h1", "type": "text"},
                "price": {"selector": ".price", "type": "number"},
                "image": {
                    "selector": "img.product",
                    "type": "attribute",
                    "attribute": "src"
                }
            }
        }
    },
)
print(response.json())
Go
package main

import (
	"bytes"
	"encoding/json"
	"fmt"
	"io"
	"net/http"
)

func main() {
	body, _ := json.Marshal(map[string]any{
		"url": "https://example.com/product/42",
		"extract": map[string]any{
			"fields": map[string]any{
				"name": map[string]any{"selector": "h1", "type": "text"},
				"price": map[string]any{"selector": ".price", "type": "number"},
				"image": map[string]any{
					"selector": "img.product",
					"type": "attribute",
					"attribute": "src",
				},
			},
		},
	})
	req, _ := http.NewRequest("POST", "https://api.domscout.io/scrape", bytes.NewReader(body))
	req.Header.Set("Content-Type", "application/json")
	req.Header.Set("x-api-key", "YOUR_API_KEY")

	res, err := http.DefaultClient.Do(req)
	if err != nil {
		panic(err)
	}
	defer res.Body.Close()

	raw, _ := io.ReadAll(res.Body)
	fmt.Println(string(raw))
}
PHP
<?php
$ch = curl_init('https://api.domscout.io/scrape');
curl_setopt_array($ch, [
    CURLOPT_POST => true,
    CURLOPT_RETURNTRANSFER => true,
    CURLOPT_HTTPHEADER => ['Content-Type: application/json', 'x-api-key: YOUR_API_KEY'],
    CURLOPT_POSTFIELDS => json_encode([
        'url' => 'https://example.com/product/42',
        'extract' => [
            'fields' => [
                'name' => ['selector' => 'h1', 'type' => 'text'],
                'price' => ['selector' => '.price', 'type' => 'number'],
                'image' => [
                    'selector' => 'img.product',
                    'type' => 'attribute',
                    'attribute' => 'src'
                ]
            ]
        ]
    ]),
]);
$response = curl_exec($ch);
if ($response === false) {
    throw new RuntimeException(curl_error($ch));
}
echo $response;
Ruby
require "json"
require "net/http"

response = Net::HTTP.post(
  URI("https://api.domscout.io/scrape"),
  {
    url: "https://example.com/product/42",
    extract: {
      fields: {
        name: { selector: "h1", type: "text" },
        price: { selector: ".price", type: "number" },
        image: { selector: "img.product", type: "attribute", attribute: "src" }
      }
    }
  }.to_json,
  "Content-Type" => "application/json",
  "x-api-key" => "YOUR_API_KEY"
)
puts response.body
Java
import java.net.URI;
import java.net.http.HttpClient;
import java.net.http.HttpRequest;
import java.net.http.HttpResponse;

public class Capture {
    public static void main(String[] args) throws Exception {
        String body = """
            {
                "url": "https://example.com/product/42",
                "extract": {
                    "fields": {
                        "name": {"selector": "h1", "type": "text"},
                        "price": {"selector": ".price", "type": "number"},
                        "image": {
                            "selector": "img.product",
                            "type": "attribute",
                            "attribute": "src"
                        }
                    }
                }
            }
            """;
        HttpRequest request = HttpRequest.newBuilder(URI.create("https://api.domscout.io/scrape"))
            .header("Content-Type", "application/json")
            .header("x-api-key", "YOUR_API_KEY")
            .POST(HttpRequest.BodyPublishers.ofString(body))
            .build();
        HttpResponse<String> response = HttpClient.newHttpClient().send(request, HttpResponse.BodyHandlers.ofString());
        System.out.println(response.body());
    }
}
C#
using System.Net.Http.Json;

using var client = new HttpClient();
client.DefaultRequestHeaders.Add("x-api-key", "YOUR_API_KEY");

var response = await client.PostAsJsonAsync("https://api.domscout.io/scrape", new {
    url = "https://example.com/product/42",
    extract = new {
        fields = new {
            name = new { selector = "h1", type = "text" },
            price = new { selector = ".price", type = "number" },
            image = new {
                selector = "img.product",
                type = "attribute",
                attribute = "src"
            }
        }
    }
});
Console.WriteLine(await response.Content.ReadAsStringAsync());

Go further

Other use cases: Web access for AI agents · Website change monitoring · Web archiving and evidence · Crawling a whole site · SEO regression checks · Thumbnails and link previews · PDFs from your own HTML · Link preview checks · Responsive layout checks

Give your product a browser.

Get clean web content and visual proof into your workflow in minutes.

Extract Structured Data from Web Pages with CSS Selectors | domscout