curl --request POST \
--url https://api.hydrafetch.com/v1/web/extract \
--header 'Content-Type: application/json' \
--header 'X-API-Key: <api-key>' \
--data '
{
"urls": [
"https://example.com/products/widget",
"https://example.com/products/*"
],
"schema": {},
"prompt": "Pull the product name, price in USD, and whether it is in stock.",
"preferStructure": true,
"enableWebSearch": true,
"showSources": true,
"showConfidence": true,
"mergeEntities": true,
"maxAge": 302400000
}
'import requests
url = "https://api.hydrafetch.com/v1/web/extract"
payload = {
"urls": ["https://example.com/products/widget", "https://example.com/products/*"],
"schema": {},
"prompt": "Pull the product name, price in USD, and whether it is in stock.",
"preferStructure": True,
"enableWebSearch": True,
"showSources": True,
"showConfidence": True,
"mergeEntities": True,
"maxAge": 302400000
}
headers = {
"X-API-Key": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'X-API-Key': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
urls: ['https://example.com/products/widget', 'https://example.com/products/*'],
schema: {},
prompt: 'Pull the product name, price in USD, and whether it is in stock.',
preferStructure: true,
enableWebSearch: true,
showSources: true,
showConfidence: true,
mergeEntities: true,
maxAge: 302400000
})
};
fetch('https://api.hydrafetch.com/v1/web/extract', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.hydrafetch.com/v1/web/extract",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'urls' => [
'https://example.com/products/widget',
'https://example.com/products/*'
],
'schema' => [
],
'prompt' => 'Pull the product name, price in USD, and whether it is in stock.',
'preferStructure' => true,
'enableWebSearch' => true,
'showSources' => true,
'showConfidence' => true,
'mergeEntities' => true,
'maxAge' => 302400000
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"X-API-Key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.hydrafetch.com/v1/web/extract"
payload := strings.NewReader("{\n \"urls\": [\n \"https://example.com/products/widget\",\n \"https://example.com/products/*\"\n ],\n \"schema\": {},\n \"prompt\": \"Pull the product name, price in USD, and whether it is in stock.\",\n \"preferStructure\": true,\n \"enableWebSearch\": true,\n \"showSources\": true,\n \"showConfidence\": true,\n \"mergeEntities\": true,\n \"maxAge\": 302400000\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("X-API-Key", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.hydrafetch.com/v1/web/extract")
.header("X-API-Key", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"urls\": [\n \"https://example.com/products/widget\",\n \"https://example.com/products/*\"\n ],\n \"schema\": {},\n \"prompt\": \"Pull the product name, price in USD, and whether it is in stock.\",\n \"preferStructure\": true,\n \"enableWebSearch\": true,\n \"showSources\": true,\n \"showConfidence\": true,\n \"mergeEntities\": true,\n \"maxAge\": 302400000\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.hydrafetch.com/v1/web/extract")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["X-API-Key"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"urls\": [\n \"https://example.com/products/widget\",\n \"https://example.com/products/*\"\n ],\n \"schema\": {},\n \"prompt\": \"Pull the product name, price in USD, and whether it is in stock.\",\n \"preferStructure\": true,\n \"enableWebSearch\": true,\n \"showSources\": true,\n \"showConfidence\": true,\n \"mergeEntities\": true,\n \"maxAge\": 302400000\n}"
response = http.request(request)
puts response.read_body{
"data": {
"results": [
{
"url": "https://example.com/products/widget",
"data": {},
"error": null,
"fields": {}
}
],
"sources": [
"<string>"
],
"collection": [
{
"data": {},
"sources": [
"<string>"
]
}
]
}
}{
"success": false,
"error": {
"code": "VALIDATION_ERROR",
"message": "url must be a valid URL"
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "UNAUTHORIZED",
"message": "Missing X-API-Key header. Agents: https://hydrafetch.com/auth.md"
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "INSUFFICIENT_CREDITS",
"message": "Insufficient credits"
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "WORKSPACE_BANNED",
"message": "This workspace has been suspended and cannot make API requests."
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "NOT_FOUND",
"message": "Job not found"
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "UPSTREAM_UNREACHABLE",
"message": "The origin did not respond to any attempt we made."
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "TOO_MANY_REQUESTS",
"message": "Rate limit exceeded for this workspace and plan."
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "INTERNAL_SERVER_ERROR",
"message": "Internal server error"
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "SERVICE_UNAVAILABLE",
"message": "The summary and json formats are not available."
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}Extract structured data from URLs
Pull schema-shaped JSON out of one or many pages in a single call. An LLM maps each page onto your schema and/or prompt. Point at a single page, a list, or a crawl scope with a trailing /* wildcard; optionally let web search find extra source pages, return per-field confidence with the source passage behind each value, and merge everything into one deduplicated collection of entities. Charged per page that returns data.
curl --request POST \
--url https://api.hydrafetch.com/v1/web/extract \
--header 'Content-Type: application/json' \
--header 'X-API-Key: <api-key>' \
--data '
{
"urls": [
"https://example.com/products/widget",
"https://example.com/products/*"
],
"schema": {},
"prompt": "Pull the product name, price in USD, and whether it is in stock.",
"preferStructure": true,
"enableWebSearch": true,
"showSources": true,
"showConfidence": true,
"mergeEntities": true,
"maxAge": 302400000
}
'import requests
url = "https://api.hydrafetch.com/v1/web/extract"
payload = {
"urls": ["https://example.com/products/widget", "https://example.com/products/*"],
"schema": {},
"prompt": "Pull the product name, price in USD, and whether it is in stock.",
"preferStructure": True,
"enableWebSearch": True,
"showSources": True,
"showConfidence": True,
"mergeEntities": True,
"maxAge": 302400000
}
headers = {
"X-API-Key": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'X-API-Key': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
urls: ['https://example.com/products/widget', 'https://example.com/products/*'],
schema: {},
prompt: 'Pull the product name, price in USD, and whether it is in stock.',
preferStructure: true,
enableWebSearch: true,
showSources: true,
showConfidence: true,
mergeEntities: true,
maxAge: 302400000
})
};
fetch('https://api.hydrafetch.com/v1/web/extract', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.hydrafetch.com/v1/web/extract",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'urls' => [
'https://example.com/products/widget',
'https://example.com/products/*'
],
'schema' => [
],
'prompt' => 'Pull the product name, price in USD, and whether it is in stock.',
'preferStructure' => true,
'enableWebSearch' => true,
'showSources' => true,
'showConfidence' => true,
'mergeEntities' => true,
'maxAge' => 302400000
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"X-API-Key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.hydrafetch.com/v1/web/extract"
payload := strings.NewReader("{\n \"urls\": [\n \"https://example.com/products/widget\",\n \"https://example.com/products/*\"\n ],\n \"schema\": {},\n \"prompt\": \"Pull the product name, price in USD, and whether it is in stock.\",\n \"preferStructure\": true,\n \"enableWebSearch\": true,\n \"showSources\": true,\n \"showConfidence\": true,\n \"mergeEntities\": true,\n \"maxAge\": 302400000\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("X-API-Key", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.hydrafetch.com/v1/web/extract")
.header("X-API-Key", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"urls\": [\n \"https://example.com/products/widget\",\n \"https://example.com/products/*\"\n ],\n \"schema\": {},\n \"prompt\": \"Pull the product name, price in USD, and whether it is in stock.\",\n \"preferStructure\": true,\n \"enableWebSearch\": true,\n \"showSources\": true,\n \"showConfidence\": true,\n \"mergeEntities\": true,\n \"maxAge\": 302400000\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.hydrafetch.com/v1/web/extract")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["X-API-Key"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"urls\": [\n \"https://example.com/products/widget\",\n \"https://example.com/products/*\"\n ],\n \"schema\": {},\n \"prompt\": \"Pull the product name, price in USD, and whether it is in stock.\",\n \"preferStructure\": true,\n \"enableWebSearch\": true,\n \"showSources\": true,\n \"showConfidence\": true,\n \"mergeEntities\": true,\n \"maxAge\": 302400000\n}"
response = http.request(request)
puts response.read_body{
"data": {
"results": [
{
"url": "https://example.com/products/widget",
"data": {},
"error": null,
"fields": {}
}
],
"sources": [
"<string>"
],
"collection": [
{
"data": {},
"sources": [
"<string>"
]
}
]
}
}{
"success": false,
"error": {
"code": "VALIDATION_ERROR",
"message": "url must be a valid URL"
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "UNAUTHORIZED",
"message": "Missing X-API-Key header. Agents: https://hydrafetch.com/auth.md"
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "INSUFFICIENT_CREDITS",
"message": "Insufficient credits"
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "WORKSPACE_BANNED",
"message": "This workspace has been suspended and cannot make API requests."
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "NOT_FOUND",
"message": "Job not found"
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "UPSTREAM_UNREACHABLE",
"message": "The origin did not respond to any attempt we made."
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "TOO_MANY_REQUESTS",
"message": "Rate limit exceeded for this workspace and plan."
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "INTERNAL_SERVER_ERROR",
"message": "Internal server error"
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}{
"success": false,
"error": {
"code": "SERVICE_UNAVAILABLE",
"message": "The summary and json formats are not available."
},
"meta": {
"requestId": "019e8a3c-9f0b-7c12-88ab-1d2e3f4a5b6c"
}
}Authorizations
Body
The pages to extract from. Each must be an http(s) URL. A trailing /* marks a crawl scope: every page discovered under that path is extracted and merged into the result.
10[
"https://example.com/products/widget",
"https://example.com/products/*"
]
JSON Schema describing the shape you want back. Optional if prompt is given; when both are present the schema fixes the field names and types while the prompt guides what to pull.
Natural-language instruction for what to extract. Use with or instead of a schema.
2000"Pull the product name, price in USD, and whether it is in stock."
Preserve document structure (headings, lists, tables) over prose density when reading the page — good for listing and catalog pages. Default off.
Pull in extra source pages by web-searching your prompt, to fill fields your URLs do not cover. Requires a prompt.
Return the concrete list of URLs that were actually extracted, after any wildcard and web-search expansion. Default off.
For each field, return a confidence score and the exact source passage the value was drawn from. Default off.
Merge the per-page results into one deduplicated collection — one row per entity, with its contributing source URLs — instead of a separate result per page. Default off.
Reuse a recent capture of each page if it is younger than this many milliseconds. Omit or 0 to always fetch fresh. Capped at 7 days.
0 <= x <= 604800000Response
The extracted data.
Show child attributes
Show child attributes
Was this page helpful?