curl "$ATAI_API_URL/agents/evals/evl_01jcb0h2m6t4xr9nv3k7pdzs5y" \
-H "Authorization: Bearer $ATAI_API_KEY"
import os
import time
import requests
base_url = os.environ["ATAI_API_URL"]
api_key = os.environ["ATAI_API_KEY"]
headers = {"Authorization": f"Bearer {api_key}"}
eval_id = "evl_01jcb0h2m6t4xr9nv3k7pdzs5y"
while True:
response = requests.get(f"{base_url}/agents/evals/{eval_id}", headers=headers)
evaluation = response.json()
if evaluation["status"] in ("completed", "failed", "cancelled"):
break
time.sleep(5)
if evaluation["status"] == "completed":
headline = evaluation["metrics_report"]["primary"]
print(f"{headline['target']}.{headline['name']} = {headline['value']}")
else:
print(f"{evaluation['status']}: {evaluation.get('error')}")
const response = await fetch(
`${process.env.ATAI_API_URL}/agents/evals/evl_01jcb0h2m6t4xr9nv3k7pdzs5y`,
{
headers: {
'Authorization': `Bearer ${process.env.ATAI_API_KEY}`
}
}
);
const body = await response.json();
if (response.ok) {
console.log(`${body.name} [${body.status}]`);
if (body.metrics_report) {
const { target, name, value } = body.metrics_report.primary;
console.log(`${target}.${name} = ${value}`);
}
} else {
console.error('Error:', body.errors);
}
{
"id": "evl_01jcb0h2m6t4xr9nv3k7pdzs5y",
"name": "Pump A regression",
"org_id": "org_01jc8m5r2vq9xt4bn7h3kdzs6w",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"primary": "state.macro_f1",
"examples": [
{
"name": "site-a-morning",
"ordinal": 1,
"inputs": [
{"type": "file", "id": "file_abc123", "format": "csv", "crc32c": "AAAAAA=="}
],
"metadata": {"site": "a", "shift": "morning"}
}
],
"rendered_config": {},
"status": "completed",
"metrics_report": {
"schema_version": "v1",
"primary": {"target": "state", "name": "macro_f1", "value": 0.87},
"targets": {
"state": {
"type": "category",
"aggregate": {"macro_f1": 0.87, "accuracy": 0.91},
"class_names": ["running", "idle", "fault_bearing"],
"per_class": {
"precision": [0.93, 0.88, 0.74],
"recall": [0.96, 0.85, 0.69],
"f1": [0.94, 0.86, 0.71],
"support": [4120, 1880, 260]
},
"confusion_matrix": [
[3955, 140, 25],
[220, 1598, 62],
[45, 36, 179]
]
}
}
},
"output_artifacts": [
{
"type": "file",
"id": "file_jkl012",
"format": "ndjson",
"crc32c": "AAAAAA==",
"metadata": {"kind": "predictions", "status": "complete", "row_count": 6260}
}
],
"created_by": "usr_01jc8m4p3rt6vx9qn2h5kdzb7y",
"created_at": "2026-09-18T11:24:03Z",
"started_at": "2026-09-18T11:24:09Z",
"completed_at": "2026-09-18T11:31:52Z",
"error": null
}
{
"errors": [
{
"code": "<error_code>",
"message": "Eval not found.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
Evals
Get Eval
Retrieve one eval, its status, and its metrics report
GET
/
agents
/
evals
/
{eval_id}
curl "$ATAI_API_URL/agents/evals/evl_01jcb0h2m6t4xr9nv3k7pdzs5y" \
-H "Authorization: Bearer $ATAI_API_KEY"
import os
import time
import requests
base_url = os.environ["ATAI_API_URL"]
api_key = os.environ["ATAI_API_KEY"]
headers = {"Authorization": f"Bearer {api_key}"}
eval_id = "evl_01jcb0h2m6t4xr9nv3k7pdzs5y"
while True:
response = requests.get(f"{base_url}/agents/evals/{eval_id}", headers=headers)
evaluation = response.json()
if evaluation["status"] in ("completed", "failed", "cancelled"):
break
time.sleep(5)
if evaluation["status"] == "completed":
headline = evaluation["metrics_report"]["primary"]
print(f"{headline['target']}.{headline['name']} = {headline['value']}")
else:
print(f"{evaluation['status']}: {evaluation.get('error')}")
const response = await fetch(
`${process.env.ATAI_API_URL}/agents/evals/evl_01jcb0h2m6t4xr9nv3k7pdzs5y`,
{
headers: {
'Authorization': `Bearer ${process.env.ATAI_API_KEY}`
}
}
);
const body = await response.json();
if (response.ok) {
console.log(`${body.name} [${body.status}]`);
if (body.metrics_report) {
const { target, name, value } = body.metrics_report.primary;
console.log(`${target}.${name} = ${value}`);
}
} else {
console.error('Error:', body.errors);
}
{
"id": "evl_01jcb0h2m6t4xr9nv3k7pdzs5y",
"name": "Pump A regression",
"org_id": "org_01jc8m5r2vq9xt4bn7h3kdzs6w",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"primary": "state.macro_f1",
"examples": [
{
"name": "site-a-morning",
"ordinal": 1,
"inputs": [
{"type": "file", "id": "file_abc123", "format": "csv", "crc32c": "AAAAAA=="}
],
"metadata": {"site": "a", "shift": "morning"}
}
],
"rendered_config": {},
"status": "completed",
"metrics_report": {
"schema_version": "v1",
"primary": {"target": "state", "name": "macro_f1", "value": 0.87},
"targets": {
"state": {
"type": "category",
"aggregate": {"macro_f1": 0.87, "accuracy": 0.91},
"class_names": ["running", "idle", "fault_bearing"],
"per_class": {
"precision": [0.93, 0.88, 0.74],
"recall": [0.96, 0.85, 0.69],
"f1": [0.94, 0.86, 0.71],
"support": [4120, 1880, 260]
},
"confusion_matrix": [
[3955, 140, 25],
[220, 1598, 62],
[45, 36, 179]
]
}
}
},
"output_artifacts": [
{
"type": "file",
"id": "file_jkl012",
"format": "ndjson",
"crc32c": "AAAAAA==",
"metadata": {"kind": "predictions", "status": "complete", "row_count": 6260}
}
],
"created_by": "usr_01jc8m4p3rt6vx9qn2h5kdzb7y",
"created_at": "2026-09-18T11:24:03Z",
"started_at": "2026-09-18T11:24:09Z",
"completed_at": "2026-09-18T11:31:52Z",
"error": null
}
{
"errors": [
{
"code": "<error_code>",
"message": "Eval not found.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
Requires version 1.1.12 or later of the Archetype platform.
Overview
This endpoint returns one eval by itsevl_ id.
Fields beyond the always-present ones are populated as the eval progresses:
started_atis populated when the runner picks up the evaloutput_artifactsis populated as the run progresses. An entry whosemetadata.statusispartialholds the rows scored so far, which is also what a failed or cancelled mid-run eval leaves behind.metrics_reportis populated upon successful completion and isnulluntil thencompleted_atis populated upon successful completionerroris populated on afailedcompletion.
metrics_report answers the question “is this agent good enough?” To answer “where is this
agent bad?” page through List Eval Examples.
Request
string
required
Eval
evl_ id.Response
string
required
TypeID-encoded eval identifier (
evl_ prefix).string
required
Human label for the eval.
string
required
Organization identifier the eval belongs to.
string
required
The blueprint being evaluated.
string
required
The run’s headline, as
<target>.<objective> — resolved at creation, so this is what was
actually scored rather than what was asked for.array
required
The examples the eval was created against, as resolved: each carries the name its results are
keyed by and inputs stamped with the CRC32C of the bytes scored and the fully-resolved
ground-truth declarations.
examples records what the run actually read; that is, resolved names, the CRC32C of the
bytes scored, and fully-resolved ground-truth declarations, rather than what the request asked
for.object
required
The fully-expanded configuration this eval runs under, resolved when the eval was created. An
eval does not run the blueprint quite unchanged: it runs the whole pipeline with a scoring
stage in place of the blueprint’s own sink, and the ground-truth labels declared alongside each
input. Opaque JSON on the wire — treat the shape as informational.
string
required
Eval lifecycle status:
pending, running, completed, failed, or cancelled.Evals are always created with the status
pending. Once they’re dispatched, their state
changes to running. From there, it will eventually enter one of the three terminal states:
completed, failed, or cancelled.string
required
Subject id (
usr_... or key_...) that created this eval.string
required
Creation timestamp (date-time).
string
When the runner picked the eval up;
null before then.string
When the eval finished;
null while unfinished.object
Aggregate and per-target scoring results. Populated when the eval reaches
completed; null
while pending, running, cancelled, or failed. See Metrics
report for the format of this object.array
Refs to the files the run produced. Each
id is a data-service file id, so an artifact is
downloadable through the files API, and each carries metadata saying which artifact it is.
See Output artifact ref for the format of the objects
in this array.Populated while the run goes, not only at the end: each batch of scored rows is appended to
the predictions file and updates the entry, whose metadata.status stays partial until the
run’s final push marks it complete. An eval that failed or was cancelled mid-run keeps its
partial entry — the rows scored so far are still readable.string
Failure detail;
null unless the eval failed.Metrics report (metrics_report)
The run’s headline plus one entry per scored target.
string
required
Wire-format version. Bumped on breaking changes; additive changes (a new target type, a new
optional field) stay on the same version.
object
required
The run’s headline score. Deliberately repeats a value that also appears in the named target’s
aggregate: a reader wanting the one number the run is judged by gets it without following a
pointer into the map.target(string, required) — the target this headline is about.name(string, required) — the metric’s name, e.g.macro_f1.value(number, required) — its value.
object
required
One entry per scored target, keyed by target name. See Target
report for the format of a target
record.
Target report (metrics_report.targets.<target name>)
object
required
Every objective this run computed for the target, by name.
string
required
The target’s type. For
category — one class per scored unit — the three fields below are also present.array
required
The class vocabulary, fixing the index order of
per_class and the confusion matrix.object
required
Per-class
precision, recall, f1, and support — parallel arrays aligned to
class_names. support counts real occurrences of each class in the ground-truth stream,
independent of what the classifier predicted.array
required
Row = true class, column = predicted class, both in
class_names order.Output artifact ref (output_artifacts[])
string
required
Storage kind, e.g.
file.string
required
Data-service file id, so the artifact is downloadable through the files API.
string
Optional format hint.
string
Whole-file CRC32C checksum of the referenced bytes, base64-encoded exactly as S3 emits it.
null when the data service has no checksum for the file.object
For an artifact an eval produced:
kind (which artifact it is), status (partial or
complete), and row_count (rows in the file — how many scored rows a reader will find in it).curl "$ATAI_API_URL/agents/evals/evl_01jcb0h2m6t4xr9nv3k7pdzs5y" \
-H "Authorization: Bearer $ATAI_API_KEY"
import os
import time
import requests
base_url = os.environ["ATAI_API_URL"]
api_key = os.environ["ATAI_API_KEY"]
headers = {"Authorization": f"Bearer {api_key}"}
eval_id = "evl_01jcb0h2m6t4xr9nv3k7pdzs5y"
while True:
response = requests.get(f"{base_url}/agents/evals/{eval_id}", headers=headers)
evaluation = response.json()
if evaluation["status"] in ("completed", "failed", "cancelled"):
break
time.sleep(5)
if evaluation["status"] == "completed":
headline = evaluation["metrics_report"]["primary"]
print(f"{headline['target']}.{headline['name']} = {headline['value']}")
else:
print(f"{evaluation['status']}: {evaluation.get('error')}")
const response = await fetch(
`${process.env.ATAI_API_URL}/agents/evals/evl_01jcb0h2m6t4xr9nv3k7pdzs5y`,
{
headers: {
'Authorization': `Bearer ${process.env.ATAI_API_KEY}`
}
}
);
const body = await response.json();
if (response.ok) {
console.log(`${body.name} [${body.status}]`);
if (body.metrics_report) {
const { target, name, value } = body.metrics_report.primary;
console.log(`${target}.${name} = ${value}`);
}
} else {
console.error('Error:', body.errors);
}
{
"id": "evl_01jcb0h2m6t4xr9nv3k7pdzs5y",
"name": "Pump A regression",
"org_id": "org_01jc8m5r2vq9xt4bn7h3kdzs6w",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"primary": "state.macro_f1",
"examples": [
{
"name": "site-a-morning",
"ordinal": 1,
"inputs": [
{"type": "file", "id": "file_abc123", "format": "csv", "crc32c": "AAAAAA=="}
],
"metadata": {"site": "a", "shift": "morning"}
}
],
"rendered_config": {},
"status": "completed",
"metrics_report": {
"schema_version": "v1",
"primary": {"target": "state", "name": "macro_f1", "value": 0.87},
"targets": {
"state": {
"type": "category",
"aggregate": {"macro_f1": 0.87, "accuracy": 0.91},
"class_names": ["running", "idle", "fault_bearing"],
"per_class": {
"precision": [0.93, 0.88, 0.74],
"recall": [0.96, 0.85, 0.69],
"f1": [0.94, 0.86, 0.71],
"support": [4120, 1880, 260]
},
"confusion_matrix": [
[3955, 140, 25],
[220, 1598, 62],
[45, 36, 179]
]
}
}
},
"output_artifacts": [
{
"type": "file",
"id": "file_jkl012",
"format": "ndjson",
"crc32c": "AAAAAA==",
"metadata": {"kind": "predictions", "status": "complete", "row_count": 6260}
}
],
"created_by": "usr_01jc8m4p3rt6vx9qn2h5kdzb7y",
"created_at": "2026-09-18T11:24:03Z",
"started_at": "2026-09-18T11:24:09Z",
"completed_at": "2026-09-18T11:31:52Z",
"error": null
}
{
"errors": [
{
"code": "<error_code>",
"message": "Eval not found.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
Important Notes
The run-level figure is every example pooled, never the mean of the per-example results from
List Eval Examples.
Was this page helpful?