curl "$ATAI_API_URL/agents/evals/evl_01jcb0h2m6t4xr9nv3k7pdzs5y/examples?limit=20" \
-H "Authorization: Bearer $ATAI_API_KEY"
import os
import requests
base_url = os.environ["ATAI_API_URL"]
api_key = os.environ["ATAI_API_KEY"]
headers = {"Authorization": f"Bearer {api_key}"}
eval_id = "evl_01jcb0h2m6t4xr9nv3k7pdzs5y"
cursor = None
while True:
params = {"limit": 100}
if cursor:
params["after"] = cursor
response = requests.get(f"{base_url}/agents/evals/{eval_id}/examples", headers=headers, params=params)
page = response.json()
for example in page["data"]:
if example["status"] == "failed":
print(f"{example['ordinal']:>3} {example['name']}: {example['error']}")
continue
score = example["targets"]["state"]["aggregate"]["macro_f1"]
print(f"{example['ordinal']:>3} {example['name']}: macro_f1={score}")
if not page["has_more"]:
break
cursor = page["next_cursor"]
const response = await fetch(
`${process.env.ATAI_API_URL}/agents/evals/evl_01jcb0h2m6t4xr9nv3k7pdzs5y/examples?limit=20`,
{
headers: {
'Authorization': `Bearer ${process.env.ATAI_API_KEY}`
}
}
);
const page = await response.json();
page.data.forEach(example => {
if (example.status === 'failed') {
console.error(`${example.ordinal} ${example.name}: ${example.error}`);
return;
}
const score = example.targets.state.aggregate.macro_f1;
console.log(`${example.ordinal} ${example.name}: macro_f1=${score}`);
});
{
"data": [
{
"id": "<example_result_id>",
"name": "site-a-morning",
"ordinal": 1,
"inputs": [
{"type": "file", "id": "file_abc123", "format": "csv", "crc32c": "AAAAAA=="}
],
"status": "completed",
"targets": {
"state": {
"type": "category",
"aggregate": {"macro_f1": 0.91, "accuracy": 0.94},
"class_names": ["running", "idle"],
"per_class": {
"precision": [0.95, 0.9],
"recall": [0.97, 0.86],
"f1": [0.96, 0.88],
"support": [2100, 900]
},
"confusion_matrix": [[2037, 63], [126, 774]]
}
},
"metadata": {"site": "a", "shift": "morning"},
"error": null
},
{
"id": "<example_result_id>",
"name": "site-b-idle-run",
"ordinal": 2,
"inputs": [
{"type": "file", "id": "file_def456", "format": "csv", "crc32c": "AAAAAA=="}
],
"status": "failed",
"metadata": {"site": "b", "shift": "night"},
"error": "<reason>"
}
],
"has_more": false,
"next_cursor": null,
"prev_cursor": null
}
{
"data": [],
"has_more": false,
"next_cursor": null,
"prev_cursor": null
}
{
"errors": [
{
"code": "<error_code>",
"message": "Invalid query parameter.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
{
"errors": [
{
"code": "<error_code>",
"message": "Eval not found.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
Evals
List Eval Examples
Page through what an eval measured for each of its examples
GET
/
agents
/
evals
/
{eval_id}
/
examples
curl "$ATAI_API_URL/agents/evals/evl_01jcb0h2m6t4xr9nv3k7pdzs5y/examples?limit=20" \
-H "Authorization: Bearer $ATAI_API_KEY"
import os
import requests
base_url = os.environ["ATAI_API_URL"]
api_key = os.environ["ATAI_API_KEY"]
headers = {"Authorization": f"Bearer {api_key}"}
eval_id = "evl_01jcb0h2m6t4xr9nv3k7pdzs5y"
cursor = None
while True:
params = {"limit": 100}
if cursor:
params["after"] = cursor
response = requests.get(f"{base_url}/agents/evals/{eval_id}/examples", headers=headers, params=params)
page = response.json()
for example in page["data"]:
if example["status"] == "failed":
print(f"{example['ordinal']:>3} {example['name']}: {example['error']}")
continue
score = example["targets"]["state"]["aggregate"]["macro_f1"]
print(f"{example['ordinal']:>3} {example['name']}: macro_f1={score}")
if not page["has_more"]:
break
cursor = page["next_cursor"]
const response = await fetch(
`${process.env.ATAI_API_URL}/agents/evals/evl_01jcb0h2m6t4xr9nv3k7pdzs5y/examples?limit=20`,
{
headers: {
'Authorization': `Bearer ${process.env.ATAI_API_KEY}`
}
}
);
const page = await response.json();
page.data.forEach(example => {
if (example.status === 'failed') {
console.error(`${example.ordinal} ${example.name}: ${example.error}`);
return;
}
const score = example.targets.state.aggregate.macro_f1;
console.log(`${example.ordinal} ${example.name}: macro_f1=${score}`);
});
{
"data": [
{
"id": "<example_result_id>",
"name": "site-a-morning",
"ordinal": 1,
"inputs": [
{"type": "file", "id": "file_abc123", "format": "csv", "crc32c": "AAAAAA=="}
],
"status": "completed",
"targets": {
"state": {
"type": "category",
"aggregate": {"macro_f1": 0.91, "accuracy": 0.94},
"class_names": ["running", "idle"],
"per_class": {
"precision": [0.95, 0.9],
"recall": [0.97, 0.86],
"f1": [0.96, 0.88],
"support": [2100, 900]
},
"confusion_matrix": [[2037, 63], [126, 774]]
}
},
"metadata": {"site": "a", "shift": "morning"},
"error": null
},
{
"id": "<example_result_id>",
"name": "site-b-idle-run",
"ordinal": 2,
"inputs": [
{"type": "file", "id": "file_def456", "format": "csv", "crc32c": "AAAAAA=="}
],
"status": "failed",
"metadata": {"site": "b", "shift": "night"},
"error": "<reason>"
}
],
"has_more": false,
"next_cursor": null,
"prev_cursor": null
}
{
"data": [],
"has_more": false,
"next_cursor": null,
"prev_cursor": null
}
{
"errors": [
{
"code": "<error_code>",
"message": "Invalid query parameter.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
{
"errors": [
{
"code": "<error_code>",
"message": "Eval not found.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
Requires version 1.1.12 or later of the Archetype platform.
Overview
This endpoint lists what an eval measured for each of its examples. The eval’s ownmetrics_report answers “is this agent good enough”; these answer “where is it
bad”. Each example is scored on its own rows as the run streams, and the run-level report is
those examples pooled — so the two always agree, and a run-level figure is never the mean of what
this returns.
Results are ordered by the example’s position in the run, oldest first. That is deliberately not
the newest-first order the other lists use: an example set has no time order, so the useful order
is the one the caller supplied.
An eval that has not completed lists no examples rather than erroring. Results appear when the
run reports them.
The HTTP status code returned is
404 both if the specified eval ID is unknown and if the ID
exists but belongs to another organization.Request
string
required
Eval
evl_ id.integer
default:"100"
Page size. Minimum
1, maximum 1000.string
Forward cursor: return examples after this page’s last one. Pass the previous page’s
next_cursor. Mutually exclusive with before.string
Backward cursor: return examples before this page’s first one. Pass the current page’s
prev_cursor. Mutually exclusive with after.Response
array
required
The page, in the order the eval was given its examples. Unlike every other list in this API
this is not newest-first: an example set has no time order, and reversing the list the caller
supplied would only make it harder to read.Each entry in this array is an Example result object.
boolean
required
true when more results exist beyond this page in the direction of travel.string
Cursor for the next page in the same direction — pass it as
after when paging forward, or as
before when you supplied before. null when has_more is false.string
Cursor to step back the way this page was reached.
null on the first page.Example result object
string
required
Identifier of this example result.
string
required
How the example is identified — its own
name, or the stem of its input file when it declared none.integer
required
The example’s 1-based position in the run, and the order results are listed in.
array
required
The example’s input refs, each carrying the CRC32C of the bytes scored.
string
required
Whether this example was scored:
completed or failed. Narrower than the eval’s own status
on purpose — an example is not dispatched, paused, or cancelled on its own. It is part of a run
that either got far enough to score it or did not.object
One entry per scored target. Absent when the example was not scored. Same shape as the run
report’s
targets, narrowed to this example’s rows — so a client that renders the run’s
numbers renders an example’s with no second code path.targets is absent on an example whose status is failed; when the status is failed,
read error instead.object
The free-form tags the example was created with. What a long per-example table is sliced by —
which site, which shift, which operator.
string
Why the example was not scored. Present when
status is failed.curl "$ATAI_API_URL/agents/evals/evl_01jcb0h2m6t4xr9nv3k7pdzs5y/examples?limit=20" \
-H "Authorization: Bearer $ATAI_API_KEY"
import os
import requests
base_url = os.environ["ATAI_API_URL"]
api_key = os.environ["ATAI_API_KEY"]
headers = {"Authorization": f"Bearer {api_key}"}
eval_id = "evl_01jcb0h2m6t4xr9nv3k7pdzs5y"
cursor = None
while True:
params = {"limit": 100}
if cursor:
params["after"] = cursor
response = requests.get(f"{base_url}/agents/evals/{eval_id}/examples", headers=headers, params=params)
page = response.json()
for example in page["data"]:
if example["status"] == "failed":
print(f"{example['ordinal']:>3} {example['name']}: {example['error']}")
continue
score = example["targets"]["state"]["aggregate"]["macro_f1"]
print(f"{example['ordinal']:>3} {example['name']}: macro_f1={score}")
if not page["has_more"]:
break
cursor = page["next_cursor"]
const response = await fetch(
`${process.env.ATAI_API_URL}/agents/evals/evl_01jcb0h2m6t4xr9nv3k7pdzs5y/examples?limit=20`,
{
headers: {
'Authorization': `Bearer ${process.env.ATAI_API_KEY}`
}
}
);
const page = await response.json();
page.data.forEach(example => {
if (example.status === 'failed') {
console.error(`${example.ordinal} ${example.name}: ${example.error}`);
return;
}
const score = example.targets.state.aggregate.macro_f1;
console.log(`${example.ordinal} ${example.name}: macro_f1=${score}`);
});
{
"data": [
{
"id": "<example_result_id>",
"name": "site-a-morning",
"ordinal": 1,
"inputs": [
{"type": "file", "id": "file_abc123", "format": "csv", "crc32c": "AAAAAA=="}
],
"status": "completed",
"targets": {
"state": {
"type": "category",
"aggregate": {"macro_f1": 0.91, "accuracy": 0.94},
"class_names": ["running", "idle"],
"per_class": {
"precision": [0.95, 0.9],
"recall": [0.97, 0.86],
"f1": [0.96, 0.88],
"support": [2100, 900]
},
"confusion_matrix": [[2037, 63], [126, 774]]
}
},
"metadata": {"site": "a", "shift": "morning"},
"error": null
},
{
"id": "<example_result_id>",
"name": "site-b-idle-run",
"ordinal": 2,
"inputs": [
{"type": "file", "id": "file_def456", "format": "csv", "crc32c": "AAAAAA=="}
],
"status": "failed",
"metadata": {"site": "b", "shift": "night"},
"error": "<reason>"
}
],
"has_more": false,
"next_cursor": null,
"prev_cursor": null
}
{
"data": [],
"has_more": false,
"next_cursor": null,
"prev_cursor": null
}
{
"errors": [
{
"code": "<error_code>",
"message": "Invalid query parameter.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
{
"errors": [
{
"code": "<error_code>",
"message": "Eval not found.",
"suggestion": null,
"error_uid": "err-xxxxxxxx"
}
]
}
Was this page helpful?