curl "$ATAI_API_URL/agents/evals?limit=20" \
-H "Authorization: Bearer $ATAI_API_KEY"
curl "$ATAI_API_URL/agents/evals?status=completed" \
-H "Authorization: Bearer $ATAI_API_KEY"
curl "$ATAI_API_URL/agents/evals?blueprint_id=blp_01jc9n7k3xf8mbq2v5t0ary6de" \
-H "Authorization: Bearer $ATAI_API_KEY"
import os
import requests
base_url = os.environ["ATAI_API_URL"]
api_key = os.environ["ATAI_API_KEY"]
headers = {"Authorization": f"Bearer {api_key}"}
cursor = None
while True:
params = {"limit": 100, "status": "completed"}
if cursor:
params["after"] = cursor
response = requests.get(f"{base_url}/agents/evals", headers=headers, params=params)
page = response.json()
for evaluation in page["data"]:
report = evaluation.get("metrics_report")
headline = report["primary"]["value"] if report else None
print(f"{evaluation['id']} {evaluation['name']} [{evaluation['status']}] {evaluation['primary']}={headline}")
if not page["has_more"]:
break
cursor = page["next_cursor"]
const params = new URLSearchParams({ limit: '20', status: 'completed' });
const response = await fetch(
`${process.env.ATAI_API_URL}/agents/evals?${params}`,
{
headers: {
'Authorization': `Bearer ${process.env.ATAI_API_KEY}`
}
}
);
const page = await response.json();
page.data.forEach(evaluation => {
const headline = evaluation.metrics_report?.primary?.value ?? null;
console.log(`${evaluation.id} ${evaluation.name} [${evaluation.status}] ${evaluation.primary}=${headline}`);
});
{
"data": [
{
"id": "evl_01jcb0h2m6t4xr9nv3k7pdzs5y",
"name": "Pump A regression",
"org_id": "org_01jc8m5r2vq9xt4bn7h3kdzs6w",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"primary": "state.macro_f1",
"examples": [
{
"name": "site-a-morning",
"ordinal": 1,
"inputs": [
{"type": "file", "id": "file_abc123", "format": "csv", "crc32c": "AAAAAA=="}
]
}
],
"rendered_config": {},
"status": "completed",
"metrics_report": {
"schema_version": "v1",
"primary": {"target": "state", "name": "macro_f1", "value": 0.87},
"targets": {
"state": {
"type": "category",
"aggregate": {"macro_f1": 0.87, "accuracy": 0.91},
"class_names": ["running", "idle"],
"per_class": {
"precision": [0.93, 0.88],
"recall": [0.96, 0.85],
"f1": [0.94, 0.86],
"support": [4120, 1880]
},
"confusion_matrix": [[3955, 165], [282, 1598]]
}
}
},
"output_artifacts": [
{
"type": "file",
"id": "file_jkl012",
"format": "ndjson",
"metadata": {"kind": "predictions", "status": "complete", "row_count": 6000}
}
],
"created_by": "usr_01jc8m4p3rt6vx9qn2h5kdzb7y",
"created_at": "2026-09-18T11:24:03Z",
"started_at": "2026-09-18T11:24:09Z",
"completed_at": "2026-09-18T11:31:52Z",
"error": null
}
],
"has_more": false,
"next_cursor": null
}
{
"data": [],
"has_more": false,
"next_cursor": null
}
{
"errors": [
{
"code": "<error_code>",
"message": "Invalid query parameter.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
Evals
List Evals
Page through evals, filtered by status or blueprint
GET
/
agents
/
evals
curl "$ATAI_API_URL/agents/evals?limit=20" \
-H "Authorization: Bearer $ATAI_API_KEY"
curl "$ATAI_API_URL/agents/evals?status=completed" \
-H "Authorization: Bearer $ATAI_API_KEY"
curl "$ATAI_API_URL/agents/evals?blueprint_id=blp_01jc9n7k3xf8mbq2v5t0ary6de" \
-H "Authorization: Bearer $ATAI_API_KEY"
import os
import requests
base_url = os.environ["ATAI_API_URL"]
api_key = os.environ["ATAI_API_KEY"]
headers = {"Authorization": f"Bearer {api_key}"}
cursor = None
while True:
params = {"limit": 100, "status": "completed"}
if cursor:
params["after"] = cursor
response = requests.get(f"{base_url}/agents/evals", headers=headers, params=params)
page = response.json()
for evaluation in page["data"]:
report = evaluation.get("metrics_report")
headline = report["primary"]["value"] if report else None
print(f"{evaluation['id']} {evaluation['name']} [{evaluation['status']}] {evaluation['primary']}={headline}")
if not page["has_more"]:
break
cursor = page["next_cursor"]
const params = new URLSearchParams({ limit: '20', status: 'completed' });
const response = await fetch(
`${process.env.ATAI_API_URL}/agents/evals?${params}`,
{
headers: {
'Authorization': `Bearer ${process.env.ATAI_API_KEY}`
}
}
);
const page = await response.json();
page.data.forEach(evaluation => {
const headline = evaluation.metrics_report?.primary?.value ?? null;
console.log(`${evaluation.id} ${evaluation.name} [${evaluation.status}] ${evaluation.primary}=${headline}`);
});
{
"data": [
{
"id": "evl_01jcb0h2m6t4xr9nv3k7pdzs5y",
"name": "Pump A regression",
"org_id": "org_01jc8m5r2vq9xt4bn7h3kdzs6w",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"primary": "state.macro_f1",
"examples": [
{
"name": "site-a-morning",
"ordinal": 1,
"inputs": [
{"type": "file", "id": "file_abc123", "format": "csv", "crc32c": "AAAAAA=="}
]
}
],
"rendered_config": {},
"status": "completed",
"metrics_report": {
"schema_version": "v1",
"primary": {"target": "state", "name": "macro_f1", "value": 0.87},
"targets": {
"state": {
"type": "category",
"aggregate": {"macro_f1": 0.87, "accuracy": 0.91},
"class_names": ["running", "idle"],
"per_class": {
"precision": [0.93, 0.88],
"recall": [0.96, 0.85],
"f1": [0.94, 0.86],
"support": [4120, 1880]
},
"confusion_matrix": [[3955, 165], [282, 1598]]
}
}
},
"output_artifacts": [
{
"type": "file",
"id": "file_jkl012",
"format": "ndjson",
"metadata": {"kind": "predictions", "status": "complete", "row_count": 6000}
}
],
"created_by": "usr_01jc8m4p3rt6vx9qn2h5kdzb7y",
"created_at": "2026-09-18T11:24:03Z",
"started_at": "2026-09-18T11:24:09Z",
"completed_at": "2026-09-18T11:31:52Z",
"error": null
}
],
"has_more": false,
"next_cursor": null
}
{
"data": [],
"has_more": false,
"next_cursor": null
}
{
"errors": [
{
"code": "<error_code>",
"message": "Invalid query parameter.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
Requires version 1.1.12 or later of the Archetype platform.
Overview
This endpoint returns a cursor-paginated page of evals, newest first. Each entry is a full eval, with the same fields Get Eval returns. Filter by lifecycle status, by the blueprint being evaluated, or search by name and ID.Request
integer
default:"100"
Page size. Minimum
1, maximum 1000.string
Forward cursor: return evals older than the one with this ID. Pass the
next_cursor of the
previous page (or the id of its last eval) to fetch the next page.This parameter is mutually exclusive with
before; do not include both.string
Backward cursor: return evals newer than the one with this ID. Pass the ID of the first eval
of the current page to walk back.
This parameter is mutually exclusive with
after; do not include both.string
Filter to a single lifecycle status (e.g.
pending, completed). Omit for all statuses. One of pending, running, completed, failed, cancelled.string
Case-insensitive substring match over the eval name and ID. Omit for no search filter.
To search for an exact blueprint ID, use the
blueprint_id parameter instead.string
Restrict to evals of this blueprint (
blp_ ID, exact match). Unlike query, this requires a
full match to the specified blueprint ID.Response
array
required
The page, newest first. Each entry is an eval, in the shape
Get Eval returns. Each entry carries the full eval. If
the eval has completed, it includes the
metrics_report, so you don’t need to make another
request to show scores.Results are ordered newest first in both cursor directions, so render
data as returned.boolean
required
true when more results exist beyond this page in the direction of travel.string
Cursor for the next page in the same direction; pass it as
after when paging forward, or as before when you supplied before. null when has_more is false.curl "$ATAI_API_URL/agents/evals?limit=20" \
-H "Authorization: Bearer $ATAI_API_KEY"
curl "$ATAI_API_URL/agents/evals?status=completed" \
-H "Authorization: Bearer $ATAI_API_KEY"
curl "$ATAI_API_URL/agents/evals?blueprint_id=blp_01jc9n7k3xf8mbq2v5t0ary6de" \
-H "Authorization: Bearer $ATAI_API_KEY"
import os
import requests
base_url = os.environ["ATAI_API_URL"]
api_key = os.environ["ATAI_API_KEY"]
headers = {"Authorization": f"Bearer {api_key}"}
cursor = None
while True:
params = {"limit": 100, "status": "completed"}
if cursor:
params["after"] = cursor
response = requests.get(f"{base_url}/agents/evals", headers=headers, params=params)
page = response.json()
for evaluation in page["data"]:
report = evaluation.get("metrics_report")
headline = report["primary"]["value"] if report else None
print(f"{evaluation['id']} {evaluation['name']} [{evaluation['status']}] {evaluation['primary']}={headline}")
if not page["has_more"]:
break
cursor = page["next_cursor"]
const params = new URLSearchParams({ limit: '20', status: 'completed' });
const response = await fetch(
`${process.env.ATAI_API_URL}/agents/evals?${params}`,
{
headers: {
'Authorization': `Bearer ${process.env.ATAI_API_KEY}`
}
}
);
const page = await response.json();
page.data.forEach(evaluation => {
const headline = evaluation.metrics_report?.primary?.value ?? null;
console.log(`${evaluation.id} ${evaluation.name} [${evaluation.status}] ${evaluation.primary}=${headline}`);
});
{
"data": [
{
"id": "evl_01jcb0h2m6t4xr9nv3k7pdzs5y",
"name": "Pump A regression",
"org_id": "org_01jc8m5r2vq9xt4bn7h3kdzs6w",
"blueprint_id": "blp_01jc9n7k3xf8mbq2v5t0ary6de",
"primary": "state.macro_f1",
"examples": [
{
"name": "site-a-morning",
"ordinal": 1,
"inputs": [
{"type": "file", "id": "file_abc123", "format": "csv", "crc32c": "AAAAAA=="}
]
}
],
"rendered_config": {},
"status": "completed",
"metrics_report": {
"schema_version": "v1",
"primary": {"target": "state", "name": "macro_f1", "value": 0.87},
"targets": {
"state": {
"type": "category",
"aggregate": {"macro_f1": 0.87, "accuracy": 0.91},
"class_names": ["running", "idle"],
"per_class": {
"precision": [0.93, 0.88],
"recall": [0.96, 0.85],
"f1": [0.94, 0.86],
"support": [4120, 1880]
},
"confusion_matrix": [[3955, 165], [282, 1598]]
}
}
},
"output_artifacts": [
{
"type": "file",
"id": "file_jkl012",
"format": "ndjson",
"metadata": {"kind": "predictions", "status": "complete", "row_count": 6000}
}
],
"created_by": "usr_01jc8m4p3rt6vx9qn2h5kdzb7y",
"created_at": "2026-09-18T11:24:03Z",
"started_at": "2026-09-18T11:24:09Z",
"completed_at": "2026-09-18T11:31:52Z",
"error": null
}
],
"has_more": false,
"next_cursor": null
}
{
"data": [],
"has_more": false,
"next_cursor": null
}
{
"errors": [
{
"code": "<error_code>",
"message": "Invalid query parameter.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
Was this page helpful?