curl --request POST \
--url https://api.hanji.dev/v1/extract/schema \
--header 'Content-Type: application/json' \
--header 'X-API-KEY: <api-key>' \
--data '
{
"url": "<string>",
"schema": {},
"strict": true,
"auto_schema": false,
"extract_images": true,
"include_ocr_text": false
}
'import requests
url = "https://api.hanji.dev/v1/extract/schema"
payload = {
"url": "<string>",
"schema": {},
"strict": True,
"auto_schema": False,
"extract_images": True,
"include_ocr_text": False
}
headers = {
"X-API-KEY": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'X-API-KEY': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
url: '<string>',
schema: {},
strict: true,
auto_schema: false,
extract_images: true,
include_ocr_text: false
})
};
fetch('https://api.hanji.dev/v1/extract/schema', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.hanji.dev/v1/extract/schema",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'url' => '<string>',
'schema' => [
],
'strict' => true,
'auto_schema' => false,
'extract_images' => true,
'include_ocr_text' => false
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"X-API-KEY: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.hanji.dev/v1/extract/schema"
payload := strings.NewReader("{\n \"url\": \"<string>\",\n \"schema\": {},\n \"strict\": true,\n \"auto_schema\": false,\n \"extract_images\": true,\n \"include_ocr_text\": false\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("X-API-KEY", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.hanji.dev/v1/extract/schema")
.header("X-API-KEY", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"url\": \"<string>\",\n \"schema\": {},\n \"strict\": true,\n \"auto_schema\": false,\n \"extract_images\": true,\n \"include_ocr_text\": false\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.hanji.dev/v1/extract/schema")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["X-API-KEY"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"url\": \"<string>\",\n \"schema\": {},\n \"strict\": true,\n \"auto_schema\": false,\n \"extract_images\": true,\n \"include_ocr_text\": false\n}"
response = http.request(request)
puts response.read_body{
"values": {},
"evidence": {},
"page_count": 123,
"ungrounded_fields": [
"<string>"
],
"generated_schema": {},
"usage": {
"pages": 123,
"credits": 123,
"credits_per_page": 123
},
"ocr_text": "<string>"
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>",
"input": "<unknown>",
"ctx": {}
}
]
}Fill a schema from a document by URL
Fill an arbitrary user JSON schema from a document URL, with citations.
curl --request POST \
--url https://api.hanji.dev/v1/extract/schema \
--header 'Content-Type: application/json' \
--header 'X-API-KEY: <api-key>' \
--data '
{
"url": "<string>",
"schema": {},
"strict": true,
"auto_schema": false,
"extract_images": true,
"include_ocr_text": false
}
'import requests
url = "https://api.hanji.dev/v1/extract/schema"
payload = {
"url": "<string>",
"schema": {},
"strict": True,
"auto_schema": False,
"extract_images": True,
"include_ocr_text": False
}
headers = {
"X-API-KEY": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'X-API-KEY': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
url: '<string>',
schema: {},
strict: true,
auto_schema: false,
extract_images: true,
include_ocr_text: false
})
};
fetch('https://api.hanji.dev/v1/extract/schema', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.hanji.dev/v1/extract/schema",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'url' => '<string>',
'schema' => [
],
'strict' => true,
'auto_schema' => false,
'extract_images' => true,
'include_ocr_text' => false
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"X-API-KEY: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.hanji.dev/v1/extract/schema"
payload := strings.NewReader("{\n \"url\": \"<string>\",\n \"schema\": {},\n \"strict\": true,\n \"auto_schema\": false,\n \"extract_images\": true,\n \"include_ocr_text\": false\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("X-API-KEY", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.hanji.dev/v1/extract/schema")
.header("X-API-KEY", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"url\": \"<string>\",\n \"schema\": {},\n \"strict\": true,\n \"auto_schema\": false,\n \"extract_images\": true,\n \"include_ocr_text\": false\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.hanji.dev/v1/extract/schema")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["X-API-KEY"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"url\": \"<string>\",\n \"schema\": {},\n \"strict\": true,\n \"auto_schema\": false,\n \"extract_images\": true,\n \"include_ocr_text\": false\n}"
response = http.request(request)
puts response.read_body{
"values": {},
"evidence": {},
"page_count": 123,
"ungrounded_fields": [
"<string>"
],
"generated_schema": {},
"usage": {
"pages": 123,
"credits": 123,
"credits_per_page": 123
},
"ocr_text": "<string>"
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>",
"input": "<unknown>",
"ctx": {}
}
]
}Authorizations
Body
Input to a schema-extraction request.
Given a document and an arbitrary user JSON schema, fill the schema's fields from the document and cite where each value came from.
HTTP(S) URL of the document to extract from. PHI-enabled keys must use the file-upload route instead.
JSON Schema describing the fields to extract; field descriptions are instructions the extractor follows. Required unless auto_schema is true.
What happens to a value whose citation cannot be verified against the document. true (default): the value is nulled out and its path listed in ungrounded_fields, so a fabricated value never reaches you. false: the value is kept but still flagged in ungrounded_fields.
Set true (and omit schema) to have a schema designed from the document first, then filled with the same grounded extraction. The schema used is returned in generated_schema.
Include figures from the parse stage in the extraction context. Set false to extract from text and tables only.
false (default): response unchanged. true: additionally return ocr_text — the whole parsed document as a single text string, concatenated in reading order (the same text POST /v1/parse returns as content).
Response
Successful Response
Show child attributes
Show child attributes
What this request charged, in credits (plan 078 D6).
Present only on responses whose request was actually charged — absent
(never null) on unbilled lanes (demo, legacy-PHI ledger) so pre-credits
response shapes stay byte-identical. credits = pages × credits_per_page
at the v1 card: parse 1.0, schema extract 4.0 all-in.
Show child attributes
Show child attributes