curl -X POST "$ATAI_API_URL/agents/evals" \
-H "Authorization: Bearer $ATAI_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"name": "Pump A regression",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"examples": [
{"inputs": [{"type": "file", "id": "file_abc123", "format": "csv"}]},
{"inputs": [{"type": "file", "id": "file_def456", "format": "csv"}]}
]
}'
curl -X POST "$ATAI_API_URL/agents/evals" \
-H "Authorization: Bearer $ATAI_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"name": "Pump A regression",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"primary": "state.macro_f1",
"emit_predictions": false,
"targets": {
"state": {
"objectives": ["macro_f1"],
"from": {"column": "ground_truth_state"},
"downsampling": "majority_vote"
}
},
"examples": [
{
"name": "site-a-morning",
"inputs": [{"type": "file", "id": "file_abc123", "format": "csv"}],
"metadata": {"site": "a", "shift": "morning"}
},
{
"name": "site-b-idle-run",
"inputs": [{"type": "file", "id": "file_def456", "format": "csv"}],
"ground_truth": {
"state": {"from": {"constant": "idle"}}
},
"metadata": {"site": "b", "shift": "night"}
}
]
}'
import os
import requests
base_url = os.environ["ATAI_API_URL"]
api_key = os.environ["ATAI_API_KEY"]
response = requests.post(
f"{base_url}/agents/evals",
headers={
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
},
json={
"name": "Pump A regression",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"examples": [
{"inputs": [{"type": "file", "id": "file_abc123", "format": "csv"}]},
{"inputs": [{"type": "file", "id": "file_def456", "format": "csv"}]},
],
},
)
if response.status_code == 201:
evaluation = response.json()
print(f"Created {evaluation['id']} ({evaluation['status']}) scoring {evaluation['primary']}")
else:
print(f"Error: {response.json()['errors']}")
const response = await fetch(`${process.env.ATAI_API_URL}/agents/evals`, {
method: 'POST',
headers: {
'Authorization': `Bearer ${process.env.ATAI_API_KEY}`,
'Content-Type': 'application/json'
},
body: JSON.stringify({
name: 'Pump A regression',
blueprint_id: 'blp_01jc9n7k3xf8mbq2v5t0ary6de',
examples: [
{ inputs: [{ type: 'file', id: 'file_abc123', format: 'csv' }] },
{ inputs: [{ type: 'file', id: 'file_def456', format: 'csv' }] }
]
})
});
const body = await response.json();
if (response.status === 201) {
console.log(`Created ${body.id} (${body.status}) scoring ${body.primary}`);
} else {
console.error('Error:', body.errors);
}
{
"id": "evl_01jcb0h2m6t4xr9nv3k7pdzs5y",
"name": "Pump A regression",
"org_id": "org_01jc8m5r2vq9xt4bn7h3kdzs6w",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"primary": "state.macro_f1",
"examples": [
{
"name": "file_abc123",
"ordinal": 1,
"inputs": [
{
"type": "file",
"id": "file_abc123",
"format": "csv",
"crc32c": "AAAAAA==",
"metadata": {
"targets": {
"state": {"from": {"column": "state"}, "downsampling": "last_record"}
}
}
}
]
}
],
"rendered_config": {},
"status": "pending",
"metrics_report": null,
"output_artifacts": null,
"created_by": "usr_01jc8m4p3rt6vx9qn2h5kdzb7y",
"created_at": "2026-09-18T11:24:03Z",
"started_at": null,
"completed_at": null,
"error": null
}
{
"errors": [
{
"code": "<error_code>",
"message": "Invalid request.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
{
"errors": [
{
"code": "<error_code>",
"message": "Blueprint not found.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
Evals
Create Eval
Score a blueprint against a set of labeled examples
POST
/
agents
/
evals
curl -X POST "$ATAI_API_URL/agents/evals" \
-H "Authorization: Bearer $ATAI_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"name": "Pump A regression",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"examples": [
{"inputs": [{"type": "file", "id": "file_abc123", "format": "csv"}]},
{"inputs": [{"type": "file", "id": "file_def456", "format": "csv"}]}
]
}'
curl -X POST "$ATAI_API_URL/agents/evals" \
-H "Authorization: Bearer $ATAI_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"name": "Pump A regression",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"primary": "state.macro_f1",
"emit_predictions": false,
"targets": {
"state": {
"objectives": ["macro_f1"],
"from": {"column": "ground_truth_state"},
"downsampling": "majority_vote"
}
},
"examples": [
{
"name": "site-a-morning",
"inputs": [{"type": "file", "id": "file_abc123", "format": "csv"}],
"metadata": {"site": "a", "shift": "morning"}
},
{
"name": "site-b-idle-run",
"inputs": [{"type": "file", "id": "file_def456", "format": "csv"}],
"ground_truth": {
"state": {"from": {"constant": "idle"}}
},
"metadata": {"site": "b", "shift": "night"}
}
]
}'
import os
import requests
base_url = os.environ["ATAI_API_URL"]
api_key = os.environ["ATAI_API_KEY"]
response = requests.post(
f"{base_url}/agents/evals",
headers={
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
},
json={
"name": "Pump A regression",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"examples": [
{"inputs": [{"type": "file", "id": "file_abc123", "format": "csv"}]},
{"inputs": [{"type": "file", "id": "file_def456", "format": "csv"}]},
],
},
)
if response.status_code == 201:
evaluation = response.json()
print(f"Created {evaluation['id']} ({evaluation['status']}) scoring {evaluation['primary']}")
else:
print(f"Error: {response.json()['errors']}")
const response = await fetch(`${process.env.ATAI_API_URL}/agents/evals`, {
method: 'POST',
headers: {
'Authorization': `Bearer ${process.env.ATAI_API_KEY}`,
'Content-Type': 'application/json'
},
body: JSON.stringify({
name: 'Pump A regression',
blueprint_id: 'blp_01jc9n7k3xf8mbq2v5t0ary6de',
examples: [
{ inputs: [{ type: 'file', id: 'file_abc123', format: 'csv' }] },
{ inputs: [{ type: 'file', id: 'file_def456', format: 'csv' }] }
]
})
});
const body = await response.json();
if (response.status === 201) {
console.log(`Created ${body.id} (${body.status}) scoring ${body.primary}`);
} else {
console.error('Error:', body.errors);
}
{
"id": "evl_01jcb0h2m6t4xr9nv3k7pdzs5y",
"name": "Pump A regression",
"org_id": "org_01jc8m5r2vq9xt4bn7h3kdzs6w",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"primary": "state.macro_f1",
"examples": [
{
"name": "file_abc123",
"ordinal": 1,
"inputs": [
{
"type": "file",
"id": "file_abc123",
"format": "csv",
"crc32c": "AAAAAA==",
"metadata": {
"targets": {
"state": {"from": {"column": "state"}, "downsampling": "last_record"}
}
}
}
]
}
],
"rendered_config": {},
"status": "pending",
"metrics_report": null,
"output_artifacts": null,
"created_by": "usr_01jc8m4p3rt6vx9qn2h5kdzb7y",
"created_at": "2026-09-18T11:24:03Z",
"started_at": null,
"completed_at": null,
"error": null
}
{
"errors": [
{
"code": "<error_code>",
"message": "Invalid request.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
{
"errors": [
{
"code": "<error_code>",
"message": "Blueprint not found.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
Requires version 1.1.12 or later of the Archetype platform.
Overview
This endpoint creates an evaluation: the blueprint to evaluate, the examples to score it on, and — only where the run differs from what the blueprint declares — how to score them.Only blueprints that publish a
metrics catalog support being evaluated.pending and returns immediately. Poll Get
Eval to watch it reach completed, failed, or
cancelled, and to read its metrics_report. Use List Eval
Examples to find out where it scored badly.
Only each example’s inputs is required. An example following the conventions its blueprint
declares needs nothing else: it is named after its input file, and each target’s labels are read
from a column named after the target. State a deviation once: on the run’s targets when every
example shares it, on the example itself when only one does.
Request
string
required
The blueprint to evaluate — a
blp_ id. Must exist and be visible to the caller’s organization.array
required
The examples to score. Must be non-empty.
string
Optional human label. Defaults to an empty string.
string
The run’s headline, as
<target>.<objective> — e.g. state.macro_f1. Defaults to the
blueprint’s own. Must name a target this run scores and an objective it computes for it.boolean
default:"true"
Emit the per-row predictions NDJSON artifact alongside the metrics report. The artifact is what
makes a score explainable. Set
false to skip it when the input is large enough that a
row-per-row artifact is not worth its size.object
Per-target overrides for this run, keyed by target name. Every field of an entry is optional
and falls back to what the blueprint’s target declares, so a run that wants nothing different
omits this entirely — and one that does need a change states it once here rather than on every
file.A key must name a target the blueprint declares, and
targets overrides apply to every example of the run; an example may still override them
for itself.objectives may only narrow what the
target offers, never widen it.Example object (examples[])
One thing an eval scores the agent on: the inputs handed to the agent, and the ground truth
expected of it.
array
required
The data handed to the agent for this example. Currently, only one
file reference is
permitted. Each entry is a data ref.inputs is the only property required on an example. A conventionally-shaped example writes nothing
else; it is named after its input file, and each target’s labels come from a column named
after the target.string
The name by which this example is identified in the run’s results. Defaults to the file name of its input,
without the extension. Names must be unique within a request: results are keyed by name, and
two examples sharing one cannot be told apart in a report.
object
The annotation supplied with this example, keyed by a name of your choosing: where its values
are and, where they are not recorded at the cadence the agent emits at, how they reduce.This says nothing about which target it serves; that’s done by
binding. This means the same annotation can
be written once and scored against different agents. Omit binding entirely to use the convention:
each target’s labels are a column of the input named after the target.object
Which targets each annotation serves, keyed by the annotation’s name; each value is an array of
target names. Omit it whenever the names line up: an annotation named after a target serves
that target, and a single annotation serves a single target whatever either is called.
object
Free-form tags for this example. Carried through to its results, where they are what makes a
long per-example table answer a question — which site, which shift, which operator.
Data ref (examples[].inputs[])
string
required
Storage kind, e.g.
file or dataset.string
required
Identifier within that storage kind, e.g.
file_abc123.string
Optional format hint, e.g.
csv.Ground-truth declaration (examples[].ground_truth.<name>)
Both fields of this object are optional and layer over what the run already settled, because the
two are independent statements: an annotation in a column of its own overrides from and leaves
the reduction alone; an annotation recorded at a coarser cadence than the rest of the set
overrides downsampling and leaves its location alone.
object
Where this example keeps the annotation’s values. Omitted to keep whatever the run settled on —
by default, a column named after the target it serves. One of two forms:
{"column": "state"}— the labels are a column of the input itself, one label per record. An optionalvaluesmap ({"0": "idle", "1": "running"}) maps each value as it appears in the column onto the label it means, for an input that spells its labels differently. A value the map does not cover is a read error, not a silently dropped record.{"constant": "idle"}— one label covers the whole input, and every record is scored against it.
string
How these labels collapse to the one value a scored unit is compared against:
last_record or
majority_vote. Omitted to keep the rule the target’s pairing declares.Target override (targets.<target name>)
The fields here are the scoring protocol — how the agent is measured — never what the agent
decides or where its decision lands, which only the blueprint may say.
array
Which of the target’s objectives to compute:
macro_f1, accuracy. Must be a subset of what
the target declares; an eval may narrow the list, never widen it.object
Where every input of this run keeps the target’s labels, when they all agree and disagree with
the target’s name. An input may still override it. Same two forms as a ground-truth
declaration’s
from.string
The window rule for every input of this run:
last_record or majority_vote. An input may
still override it.Response
Returns201 Created with the eval in pending status.
string
required
TypeID-encoded eval identifier (
evl_ prefix).string
required
Human label for the eval.
string
required
Organization identifier the eval belongs to.
string
required
The blueprint being evaluated.
string
required
The run’s headline, as
<target>.<objective> — resolved at creation, so this is what was
actually scored rather than what was asked for.array
required
The examples the eval was created against, as resolved: each carries the name its results are
keyed by and inputs stamped with the CRC32C of the bytes scored and the fully-resolved
ground-truth declarations, so the row records what was read rather than what was asked for.
object
required
The fully-expanded configuration this eval runs under, resolved when the eval was created: the
blueprint being evaluated, the bundle’s values and artifact locations, and the test files to
score against. A self-describing record of the run, so an eval still says exactly what it ran
even after its bundle or blueprint moves on. Opaque JSON on the wire — treat the shape as
informational.
string
required
Eval lifecycle status:
pending, running, completed, failed, or cancelled.string
required
Subject id (
usr_... or key_...) that created this eval.string
required
Creation timestamp (date-time).
string
When the runner picked the eval up;
null before then.string
When the eval finished;
null while unfinished.object
Aggregate and per-target scoring results. Populated when the eval reaches
completed; null
while pending, running, cancelled, or failed. See
Get Eval for its shape.array
Refs to the files the run produced. Each
id is a data-service file id, so an artifact is
downloadable through the files API.string
Failure detail;
null unless the eval failed.Resolved example object (examples[])
string
required
How this example is identified in the run’s results.
Example names must be unique within a request, because results are keyed by name.
integer
required
The example’s 1-based position in the run, fixing the order results are listed in.
array
required
The data handed to the agent, each ref carrying the CRC32C of the bytes scored.
object
The free-form tags the example was created with, if any.
curl -X POST "$ATAI_API_URL/agents/evals" \
-H "Authorization: Bearer $ATAI_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"name": "Pump A regression",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"examples": [
{"inputs": [{"type": "file", "id": "file_abc123", "format": "csv"}]},
{"inputs": [{"type": "file", "id": "file_def456", "format": "csv"}]}
]
}'
curl -X POST "$ATAI_API_URL/agents/evals" \
-H "Authorization: Bearer $ATAI_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"name": "Pump A regression",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"primary": "state.macro_f1",
"emit_predictions": false,
"targets": {
"state": {
"objectives": ["macro_f1"],
"from": {"column": "ground_truth_state"},
"downsampling": "majority_vote"
}
},
"examples": [
{
"name": "site-a-morning",
"inputs": [{"type": "file", "id": "file_abc123", "format": "csv"}],
"metadata": {"site": "a", "shift": "morning"}
},
{
"name": "site-b-idle-run",
"inputs": [{"type": "file", "id": "file_def456", "format": "csv"}],
"ground_truth": {
"state": {"from": {"constant": "idle"}}
},
"metadata": {"site": "b", "shift": "night"}
}
]
}'
import os
import requests
base_url = os.environ["ATAI_API_URL"]
api_key = os.environ["ATAI_API_KEY"]
response = requests.post(
f"{base_url}/agents/evals",
headers={
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
},
json={
"name": "Pump A regression",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"examples": [
{"inputs": [{"type": "file", "id": "file_abc123", "format": "csv"}]},
{"inputs": [{"type": "file", "id": "file_def456", "format": "csv"}]},
],
},
)
if response.status_code == 201:
evaluation = response.json()
print(f"Created {evaluation['id']} ({evaluation['status']}) scoring {evaluation['primary']}")
else:
print(f"Error: {response.json()['errors']}")
const response = await fetch(`${process.env.ATAI_API_URL}/agents/evals`, {
method: 'POST',
headers: {
'Authorization': `Bearer ${process.env.ATAI_API_KEY}`,
'Content-Type': 'application/json'
},
body: JSON.stringify({
name: 'Pump A regression',
blueprint_id: 'blp_01jc9n7k3xf8mbq2v5t0ary6de',
examples: [
{ inputs: [{ type: 'file', id: 'file_abc123', format: 'csv' }] },
{ inputs: [{ type: 'file', id: 'file_def456', format: 'csv' }] }
]
})
});
const body = await response.json();
if (response.status === 201) {
console.log(`Created ${body.id} (${body.status}) scoring ${body.primary}`);
} else {
console.error('Error:', body.errors);
}
{
"id": "evl_01jcb0h2m6t4xr9nv3k7pdzs5y",
"name": "Pump A regression",
"org_id": "org_01jc8m5r2vq9xt4bn7h3kdzs6w",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"primary": "state.macro_f1",
"examples": [
{
"name": "file_abc123",
"ordinal": 1,
"inputs": [
{
"type": "file",
"id": "file_abc123",
"format": "csv",
"crc32c": "AAAAAA==",
"metadata": {
"targets": {
"state": {"from": {"column": "state"}, "downsampling": "last_record"}
}
}
}
]
}
],
"rendered_config": {},
"status": "pending",
"metrics_report": null,
"output_artifacts": null,
"created_by": "usr_01jc8m4p3rt6vx9qn2h5kdzb7y",
"created_at": "2026-09-18T11:24:03Z",
"started_at": null,
"completed_at": null,
"error": null
}
{
"errors": [
{
"code": "<error_code>",
"message": "Invalid request.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
{
"errors": [
{
"code": "<error_code>",
"message": "Blueprint not found.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
Was this page helpful?