curl --request POST \
--url https://api.anyformat.ai/v2/workflows/ \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data @- <<EOF
{
"description": "Pull the invoice number and total from each document.",
"edges": [
{
"source": "parse",
"target": "extract"
}
],
"name": "Invoice totals",
"nodes": [
{
"id": "parse",
"mode": "standard",
"type": "parse"
},
{
"extraction_schema": {
"fields": [
{
"data_type": "string",
"description": "Invoice number as printed on the document.",
"name": "invoice_number"
},
{
"data_type": "float",
"description": "Invoice total in the document's currency.",
"name": "total_amount"
}
]
},
"id": "extract",
"mode": "standard",
"type": "extract"
}
]
}
EOFimport requests
url = "https://api.anyformat.ai/v2/workflows/"
payload = {
"description": "Pull the invoice number and total from each document.",
"edges": [
{
"source": "parse",
"target": "extract"
}
],
"name": "Invoice totals",
"nodes": [
{
"id": "parse",
"mode": "standard",
"type": "parse"
},
{
"extraction_schema": { "fields": [
{
"data_type": "string",
"description": "Invoice number as printed on the document.",
"name": "invoice_number"
},
{
"data_type": "float",
"description": "Invoice total in the document's currency.",
"name": "total_amount"
}
] },
"id": "extract",
"mode": "standard",
"type": "extract"
}
]
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
description: 'Pull the invoice number and total from each document.',
edges: [{source: 'parse', target: 'extract'}],
name: 'Invoice totals',
nodes: [
{id: 'parse', mode: 'standard', type: 'parse'},
{
extraction_schema: {
fields: [
{
data_type: 'string',
description: 'Invoice number as printed on the document.',
name: 'invoice_number'
},
{
data_type: 'float',
description: 'Invoice total in the document\'s currency.',
name: 'total_amount'
}
]
},
id: 'extract',
mode: 'standard',
type: 'extract'
}
]
})
};
fetch('https://api.anyformat.ai/v2/workflows/', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.anyformat.ai/v2/workflows/",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'description' => 'Pull the invoice number and total from each document.',
'edges' => [
[
'source' => 'parse',
'target' => 'extract'
]
],
'name' => 'Invoice totals',
'nodes' => [
[
'id' => 'parse',
'mode' => 'standard',
'type' => 'parse'
],
[
'extraction_schema' => [
'fields' => [
[
'data_type' => 'string',
'description' => 'Invoice number as printed on the document.',
'name' => 'invoice_number'
],
[
'data_type' => 'float',
'description' => 'Invoice total in the document\'s currency.',
'name' => 'total_amount'
]
]
],
'id' => 'extract',
'mode' => 'standard',
'type' => 'extract'
]
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.anyformat.ai/v2/workflows/"
payload := strings.NewReader("{\n \"description\": \"Pull the invoice number and total from each document.\",\n \"edges\": [\n {\n \"source\": \"parse\",\n \"target\": \"extract\"\n }\n ],\n \"name\": \"Invoice totals\",\n \"nodes\": [\n {\n \"id\": \"parse\",\n \"mode\": \"standard\",\n \"type\": \"parse\"\n },\n {\n \"extraction_schema\": {\n \"fields\": [\n {\n \"data_type\": \"string\",\n \"description\": \"Invoice number as printed on the document.\",\n \"name\": \"invoice_number\"\n },\n {\n \"data_type\": \"float\",\n \"description\": \"Invoice total in the document's currency.\",\n \"name\": \"total_amount\"\n }\n ]\n },\n \"id\": \"extract\",\n \"mode\": \"standard\",\n \"type\": \"extract\"\n }\n ]\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.anyformat.ai/v2/workflows/")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"description\": \"Pull the invoice number and total from each document.\",\n \"edges\": [\n {\n \"source\": \"parse\",\n \"target\": \"extract\"\n }\n ],\n \"name\": \"Invoice totals\",\n \"nodes\": [\n {\n \"id\": \"parse\",\n \"mode\": \"standard\",\n \"type\": \"parse\"\n },\n {\n \"extraction_schema\": {\n \"fields\": [\n {\n \"data_type\": \"string\",\n \"description\": \"Invoice number as printed on the document.\",\n \"name\": \"invoice_number\"\n },\n {\n \"data_type\": \"float\",\n \"description\": \"Invoice total in the document's currency.\",\n \"name\": \"total_amount\"\n }\n ]\n },\n \"id\": \"extract\",\n \"mode\": \"standard\",\n \"type\": \"extract\"\n }\n ]\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.anyformat.ai/v2/workflows/")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"description\": \"Pull the invoice number and total from each document.\",\n \"edges\": [\n {\n \"source\": \"parse\",\n \"target\": \"extract\"\n }\n ],\n \"name\": \"Invoice totals\",\n \"nodes\": [\n {\n \"id\": \"parse\",\n \"mode\": \"standard\",\n \"type\": \"parse\"\n },\n {\n \"extraction_schema\": {\n \"fields\": [\n {\n \"data_type\": \"string\",\n \"description\": \"Invoice number as printed on the document.\",\n \"name\": \"invoice_number\"\n },\n {\n \"data_type\": \"float\",\n \"description\": \"Invoice total in the document's currency.\",\n \"name\": \"total_amount\"\n }\n ]\n },\n \"id\": \"extract\",\n \"mode\": \"standard\",\n \"type\": \"extract\"\n }\n ]\n}"
response = http.request(request)
puts response.read_bodyCreate Workflow
Create a workflow from a typed graph of parse / classify / splitter / extract nodes in a single atomic transaction
curl --request POST \
--url https://api.anyformat.ai/v2/workflows/ \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data @- <<EOF
{
"description": "Pull the invoice number and total from each document.",
"edges": [
{
"source": "parse",
"target": "extract"
}
],
"name": "Invoice totals",
"nodes": [
{
"id": "parse",
"mode": "standard",
"type": "parse"
},
{
"extraction_schema": {
"fields": [
{
"data_type": "string",
"description": "Invoice number as printed on the document.",
"name": "invoice_number"
},
{
"data_type": "float",
"description": "Invoice total in the document's currency.",
"name": "total_amount"
}
]
},
"id": "extract",
"mode": "standard",
"type": "extract"
}
]
}
EOFimport requests
url = "https://api.anyformat.ai/v2/workflows/"
payload = {
"description": "Pull the invoice number and total from each document.",
"edges": [
{
"source": "parse",
"target": "extract"
}
],
"name": "Invoice totals",
"nodes": [
{
"id": "parse",
"mode": "standard",
"type": "parse"
},
{
"extraction_schema": { "fields": [
{
"data_type": "string",
"description": "Invoice number as printed on the document.",
"name": "invoice_number"
},
{
"data_type": "float",
"description": "Invoice total in the document's currency.",
"name": "total_amount"
}
] },
"id": "extract",
"mode": "standard",
"type": "extract"
}
]
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
description: 'Pull the invoice number and total from each document.',
edges: [{source: 'parse', target: 'extract'}],
name: 'Invoice totals',
nodes: [
{id: 'parse', mode: 'standard', type: 'parse'},
{
extraction_schema: {
fields: [
{
data_type: 'string',
description: 'Invoice number as printed on the document.',
name: 'invoice_number'
},
{
data_type: 'float',
description: 'Invoice total in the document\'s currency.',
name: 'total_amount'
}
]
},
id: 'extract',
mode: 'standard',
type: 'extract'
}
]
})
};
fetch('https://api.anyformat.ai/v2/workflows/', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.anyformat.ai/v2/workflows/",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'description' => 'Pull the invoice number and total from each document.',
'edges' => [
[
'source' => 'parse',
'target' => 'extract'
]
],
'name' => 'Invoice totals',
'nodes' => [
[
'id' => 'parse',
'mode' => 'standard',
'type' => 'parse'
],
[
'extraction_schema' => [
'fields' => [
[
'data_type' => 'string',
'description' => 'Invoice number as printed on the document.',
'name' => 'invoice_number'
],
[
'data_type' => 'float',
'description' => 'Invoice total in the document\'s currency.',
'name' => 'total_amount'
]
]
],
'id' => 'extract',
'mode' => 'standard',
'type' => 'extract'
]
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.anyformat.ai/v2/workflows/"
payload := strings.NewReader("{\n \"description\": \"Pull the invoice number and total from each document.\",\n \"edges\": [\n {\n \"source\": \"parse\",\n \"target\": \"extract\"\n }\n ],\n \"name\": \"Invoice totals\",\n \"nodes\": [\n {\n \"id\": \"parse\",\n \"mode\": \"standard\",\n \"type\": \"parse\"\n },\n {\n \"extraction_schema\": {\n \"fields\": [\n {\n \"data_type\": \"string\",\n \"description\": \"Invoice number as printed on the document.\",\n \"name\": \"invoice_number\"\n },\n {\n \"data_type\": \"float\",\n \"description\": \"Invoice total in the document's currency.\",\n \"name\": \"total_amount\"\n }\n ]\n },\n \"id\": \"extract\",\n \"mode\": \"standard\",\n \"type\": \"extract\"\n }\n ]\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.anyformat.ai/v2/workflows/")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"description\": \"Pull the invoice number and total from each document.\",\n \"edges\": [\n {\n \"source\": \"parse\",\n \"target\": \"extract\"\n }\n ],\n \"name\": \"Invoice totals\",\n \"nodes\": [\n {\n \"id\": \"parse\",\n \"mode\": \"standard\",\n \"type\": \"parse\"\n },\n {\n \"extraction_schema\": {\n \"fields\": [\n {\n \"data_type\": \"string\",\n \"description\": \"Invoice number as printed on the document.\",\n \"name\": \"invoice_number\"\n },\n {\n \"data_type\": \"float\",\n \"description\": \"Invoice total in the document's currency.\",\n \"name\": \"total_amount\"\n }\n ]\n },\n \"id\": \"extract\",\n \"mode\": \"standard\",\n \"type\": \"extract\"\n }\n ]\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.anyformat.ai/v2/workflows/")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"description\": \"Pull the invoice number and total from each document.\",\n \"edges\": [\n {\n \"source\": \"parse\",\n \"target\": \"extract\"\n }\n ],\n \"name\": \"Invoice totals\",\n \"nodes\": [\n {\n \"id\": \"parse\",\n \"mode\": \"standard\",\n \"type\": \"parse\"\n },\n {\n \"extraction_schema\": {\n \"fields\": [\n {\n \"data_type\": \"string\",\n \"description\": \"Invoice number as printed on the document.\",\n \"name\": \"invoice_number\"\n },\n {\n \"data_type\": \"float\",\n \"description\": \"Invoice total in the document's currency.\",\n \"name\": \"total_amount\"\n }\n ]\n },\n \"id\": \"extract\",\n \"mode\": \"standard\",\n \"type\": \"extract\"\n }\n ]\n}"
response = http.request(request)
puts response.read_bodyPOST /v2/workflows/ creates a workflow from a strongly-typed graph. Use it to:
- Configure parse-node settings: standard or agentic mode, prompt hints, and figure enhancement.
- Build a parse-only workflow, with no extract node, that returns markdown only.
- Route documents through a classifier or a splitter to several extract nodes.
- Run a linear
parse → extractworkflow.
Request Body
| Field | Type | Required | Description |
|---|---|---|---|
name | string | Yes | Workflow name |
description | string | No | Optional description |
nodes | Node[] | Yes | At least one node. Exactly one must be type="parse" |
edges | Edge[] | No | Directed edges. Empty for a parse-only workflow |
Parse Node
The entry-point node. Exactly one per workflow.| Field | Type | Default | Description |
|---|---|---|---|
id | string | none | Stable identifier, such as "parse_1" |
type | "parse" | none | Discriminator |
mode | "standard" | "agentic" | "standard" | Use "agentic" for per-block LLM routing through the typed text, table and figure strategies |
prompt_hint | string | null | Domain hint shown to the parser, such as “medical lab report, preserve numerics exactly” |
figure_enhancement | boolean | false | Extract structured descriptions of charts and images. Costs extra LLM spend |
Extract Node
Pulls structured fields from upstream parsed content.| Field | Type | Required | Description |
|---|---|---|---|
id | string | Yes | Stable identifier |
type | "extract" | Yes | Discriminator |
extraction_schema | object | Yes | { "fields": [...] }, with at least one field |
use_images | boolean | No (default false) | Pass rendered page images to the extraction model |
lookup_file_uploads | object[] | No | Inline reference files for smart lookup, each { "filename": string, "content": "<base64>" } |
lookup_file_ids | string[] | No | Ids of reference files already attached to the workflow, as a read returns them. A new workflow has none, so attach files through lookup_file_uploads |
lookup_suggestion | string | No | Free-form hint telling the matcher how to join the document against the reference file |
"source": "smart_lookup" to become a smart-lookup field, resolved from a reference file rather than read from the document. "source": "lookup_if_missing" extracts the field too, and lets the lookup fill only the empty slots. The default is "extraction".
Classify Node
Routes the document to one ofcategories[] based on an LLM verdict. Outgoing edges must set branch to a category id.
| Field | Type | Required | Description |
|---|---|---|---|
id | string | Yes | Stable identifier |
type | "classify" | Yes | Discriminator |
categories | Category[] | Yes | At least one category |
user_prompt | string | No | Optional prompt prefix shown to the classifier |
Category:
| Field | Type | Required | Description |
|---|---|---|---|
id | string | Yes | Stable category id; matched against an outgoing edge’s branch value |
name | string | Yes | Display name shown to the LLM |
description | string | Yes | Free-form description shown to the LLM |
Splitter Node
Partitions a multi-document file into per-rule sub-documents. Each rule fires an outgoing edge whosebranch matches the rule id. The downstream extract node runs once per resulting partition.
| Field | Type | Required | Description |
|---|---|---|---|
id | string | Yes | Stable identifier |
type | "splitter" | Yes | Discriminator |
rules | Rule[] | Yes | At least one rule |
Rule:
| Field | Type | Required | Description |
|---|---|---|---|
id | string | Yes | Stable rule id; matched against an outgoing edge’s branch value |
name | string | Yes | Display name |
description | string | Yes | Free-form description shown to the splitter model |
partition_key | string | No, default "" | Key used to label the resulting partitions. Empty means routing only |
Validate Node
Runs after anextract node and emits per-rule validation results. The results endpoint returns them as extractions[].validations[], beside the fields they judged. See Response Formats. A rule is either AI-evaluated or deterministic. An AI rule sets kind="ai" and carries a natural-language description. A deterministic rule sets kind="deterministic" and carries a structured check, evaluated in pure Python with no model call.
| Field | Type | Required | Description |
|---|---|---|---|
id | string | Yes | Stable identifier |
type | "validate" | Yes | Discriminator |
rules | Rule[] | Yes | At least one rule |
Rule:
| Field | Type | Required | Description |
|---|---|---|---|
id | string | Yes | Stable rule id. Round-trips through ValidationResult.rule_id |
kind | "ai" | "deterministic" | No, default "ai" | Evaluation strategy |
name | string | No | Display name for the rule card |
description | string | Conditional | Natural-language prompt. Required when kind="ai", forbidden when kind="deterministic" |
check | Check | Conditional | Structured deterministic check. Required when kind="deterministic", forbidden when kind="ai" |
severity | "error" | "warning" | No, default "error" | Severity of a violation |
source_fields | string[] | No | persistent_ids of fields the rule references |
check shapes are discriminated on type. Every shape except expression names its field operands by persistent_id. expression addresses fields by extracted name instead.
type | Required fields | Optional fields | Meaning |
|---|---|---|---|
range | field | min, max, at least one | Numeric field is within [min, max] |
date | field | earliest, latest, as ISO YYYY-MM-DD or the literal "today" | Date field is within the inclusive bounds |
arithmetic | operands[], at least one item, and equals | operator, one of "sum" | "subtract" | "product" and defaulting to "sum", plus tolerance, defaulting to 0.0 | operator(operands) equals the value of the equals field, within tolerance |
comparison | left, op, right | none | left <op> right holds. op is "==" | "!=" | ">" | ">=" | "<" | "<=". right is {"source": "field", "field": "<pid>"} or {"source": "literal", "value": <number|string|bool>} |
one_of | field, allowed[], at least one item | case_sensitive, defaulting to false | field’s value is one of allowed |
regex | field, pattern | none | field matches pattern. google-re2 evaluates it in linear time, and rejects stdlib re syntax that is not re2-safe |
required | field | none | field is present, meaning non-null and non-empty |
expression | expression, 1–1000 chars | none | A CEL expression over the whole extraction, bound to one root variable data. Fields are addressed by extracted name, as in data.total or data.lineas.map(l, l.importe). CEL gains the helpers num(v), sum(list) and abs(x). Anything the evaluator cannot answer with a boolean is inconclusive |
subtotal + tax == total, within ±1¢:
{
"id": "totals_balance",
"kind": "deterministic",
"check": {
"type": "arithmetic",
"operator": "sum",
"operands": ["pid_subtotal", "pid_tax"],
"equals": "pid_total",
"tolerance": 0.01
},
"source_fields": ["pid_subtotal", "pid_tax", "pid_total"]
}
Edges
| Field | Type | Required | Description |
|---|---|---|---|
source | string | Yes | Source node id |
target | string | Yes | Target node id |
branch | string | Conditional | Required when leaving a classify or splitter node, and forbidden otherwise. It must equal a category or rule .id on the source node, never its .name. Using .name is rejected with 400. |
Examples
Linear parse → extract
curl -X POST 'https://api.anyformat.ai/v2/workflows/' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer YOUR_API_KEY' \
-d '{
"name": "Invoice extractor",
"nodes": [
{"id": "parse_1", "type": "parse"},
{
"id": "extract_1",
"type": "extract",
"extraction_schema": {
"fields": [
{"name": "invoice_number", "description": "Invoice ID", "data_type": "string"},
{"name": "total", "description": "Grand total", "data_type": "float"}
]
}
}
],
"edges": [{"source": "parse_1", "target": "extract_1"}]
}'
import requests
response = requests.post(
"https://api.anyformat.ai/v2/workflows/",
headers={"Authorization": "Bearer YOUR_API_KEY"},
json={
"name": "Invoice extractor",
"nodes": [
{"id": "parse_1", "type": "parse"},
{
"id": "extract_1",
"type": "extract",
"extraction_schema": {
"fields": [
{"name": "invoice_number", "description": "Invoice ID", "data_type": "string"},
{"name": "total", "description": "Grand total", "data_type": "float"},
]
},
},
],
"edges": [{"source": "parse_1", "target": "extract_1"}],
},
)
workflow_id = response.json()["id"]
Parse-only with agentic mode
A workflow with oneparse node and no edges. It produces markdown only, and extracts nothing.
curl -X POST 'https://api.anyformat.ai/v2/workflows/' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer YOUR_API_KEY' \
-d '{
"name": "Agentic parse-only",
"nodes": [
{
"id": "parse_1",
"type": "parse",
"mode": "agentic"
}
],
"edges": []
}'
import requests
response = requests.post(
"https://api.anyformat.ai/v2/workflows/",
headers={"Authorization": "Bearer YOUR_API_KEY"},
json={
"name": "Agentic parse-only",
"nodes": [
{
"id": "parse_1",
"type": "parse",
"mode": "agentic",
}
],
"edges": [],
},
)
workflow_id = response.json()["id"]
mode: "agentic" turns on the full agentic pipeline. See the Agentic Parse to Markdown recipe for an end-to-end walkthrough.Classify-then-extract (branched)
Each outgoing edge from aclassify or splitter node sets branch to a category or rule .id on the source node. .name is the display label shown to the LLM, and is not a valid branch value: using it returns 400. In the example below, cat_invoice routes the edge, and Invoice is the label the classifier sees.
curl -X POST 'https://api.anyformat.ai/v2/workflows/' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer YOUR_API_KEY' \
-d '{
"name": "Multi-doc",
"nodes": [
{"id": "parse_1", "type": "parse"},
{
"id": "classify_1",
"type": "classify",
"categories": [
{"id": "cat_invoice", "name": "Invoice", "description": "Vendor invoice."},
{"id": "cat_receipt", "name": "Receipt", "description": "POS receipt."}
]
},
{
"id": "extract_invoice",
"type": "extract",
"extraction_schema": {"fields": [{"name": "vendor", "description": "Vendor name.", "data_type": "string"}]}
},
{
"id": "extract_receipt",
"type": "extract",
"extraction_schema": {"fields": [{"name": "merchant", "description": "Merchant name.", "data_type": "string"}]}
}
],
"edges": [
{"source": "parse_1", "target": "classify_1"},
{"source": "classify_1", "target": "extract_invoice", "branch": "cat_invoice"},
{"source": "classify_1", "target": "extract_receipt", "branch": "cat_receipt"}
]
}'
Smart lookup
Enrich extracted values by matching them against a reference file instead of reading them off the page. Flag the looked-up field with"source": "smart_lookup". Attach the reference file inline as base64 through lookup_file_uploads. Describe the join with lookup_suggestion if it helps. See Smart Lookup for how matching works and for the Studio equivalent of these fields. In the example below, the model reads vendor_name from the document, then resolves the canonical vendor_id from a vendor catalog.
curl -X POST 'https://api.anyformat.ai/v2/workflows/' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer YOUR_API_KEY' \
-d '{
"name": "Invoice + vendor lookup",
"nodes": [
{"id": "parse_1", "type": "parse"},
{
"id": "extract_1",
"type": "extract",
"lookup_file_uploads": [
{"filename": "vendor_catalog.csv", "content": "dmVuZG9yX25hbWUsdmVuZG9yX2lkCkFjbWUgQ29ycCxWLTAwMDEK"}
],
"lookup_suggestion": "Match the extracted vendor_name against the vendor_name column and return the matching vendor_id.",
"extraction_schema": {
"fields": [
{"name": "vendor_name", "description": "Vendor as printed on the document", "data_type": "string"},
{"name": "total", "description": "Grand total", "data_type": "float"},
{"name": "vendor_id", "description": "Canonical vendor code from the catalog, joined on vendor_name", "data_type": "string", "source": "smart_lookup"}
]
}
}
],
"edges": [{"source": "parse_1", "target": "extract_1"}]
}'
lookup_file_uploads[].content is the base64-encoded bytes of the reference CSV or text file. The results payload returns the lookup field, vendor_id, alongside the extraction fields. It has no separate response section. An extract node carrying a source: smart_lookup or lookup_if_missing field but no reference file is rejected with 400.Topology Rules
The endpoint enforces graph correctness. A violated rule returns400, and the API persists nothing.
| Rule | Message |
|---|---|
| Exactly one parse node | workflow must contain exactly one \parse` node` |
| Unique node ids | duplicate node ids: [...] |
| Edges reference existing nodes | edge source X not in nodes or edge target Y not in nodes |
| Edge predecessor compatibility | extract does not accept extract as predecessor |
| Branch routing | edge from classify X requires \branch` to be set` |
Branch matches source .id | edge from classify 'classify_1' sets branch='Invoice', but no category on 'classify_1' has that id (valid ids: ['cat_invoice']). \branch` must match a category `.id`, not its `.name`.` |
| No fan-out from non-routers | node Y (extract) has 3 outgoing edges; only classify and splitter may fan out |
Response
201 Created returns the workflow resource:
{
"id": "0686bb97-8c30-70f0-8000-97669e000eb8",
"name": "Invoice extractor",
"description": "",
"created_at": "2026-05-11T10:00:00Z",
"updated_at": "2026-05-11T10:00:00Z"
}
id. You reuse it when uploading documents and fetching results.
Next Steps
Agentic Parse to Markdown
Parse-Only Workflow
Field Types
data_type values for extraction fieldsResponse Formats
parse, extractions, splits and classifications on the results endpointAuthorizations
API key issued from app.anyformat.ai/api-key. Send as Authorization: Bearer <key>.
Body
Public-surface workflow create body — typed graph of parse / classify / splitter / extract / validate nodes.
A validate node carries rules; each rule is either AI-evaluated
(kind="ai" + a natural-language description) or deterministic
(kind="deterministic" + a structured check, evaluated in pure
Python with no model call).
Uses PublicNode (public request models with only the fields callers
may set); the domain AnyNode also carries filter (worker-only)
and executor-injected / staff-only fields, so the public bodies pin a
stricter node schema. Call :meth:to_domain before forwarding to the
backend.
1"Invoice or receipt"
1- PublicParseNode
- ClassifyNode
- SplitterNode
- PublicExtractNode
- PublicValidateNode
- PublicIfElseNode
- PublicSlackAlertNode
- PublicEditNode
- KnowledgeNode
Show child attributes
Show child attributes
Show child attributes
Show child attributes
Response
Successful Response
A workflow defines the extraction template — what fields to extract from documents, their types, and validation rules.
Unique identifier of the workflow (UUID).
"0686bb97-8c30-70f0-8000-97669e000eb8"
Human-readable name of the workflow.
"Invoice Processing"
Optional description of what this workflow extracts.
"A workflow for processing invoices and retrieving invoice details."
Timestamp when the workflow was created (ISO 8601).
Timestamp when the workflow was last modified (ISO 8601).
List of extraction field definitions configured for this workflow. null if not yet configured.

