← Files AMDARCHIVED FILE
eval/TEMPLATE.json
1.39 KB · Sep 30, 2026 · 23:13 UTC
{
"evaluations": [
{
"id": "REPLACE-positive-with-expectations",
"skill_should_trigger": true,
"note": "Tier 1. The same prompt grades routing AND what the agent then did.",
"prompt": "A task this skill should carry out end to end.",
"expected_behavior": [
"A thing the agent must do, in plain language, graded by an LLM judge"
],
"unexpected_behavior": [
"A mistake this skill exists to prevent"
],
"logs_contain": [
"a-literal-that-must-appear.py"
],
"files_exist": [
"expected-output.txt"
]
},
{
"id": "REPLACE-unique-slug",
"skill_should_trigger": true,
"prompt": "Something a user would say that should make this skill fire."
},
{
"id": "REPLACE-second-positive",
"skill_should_trigger": true,
"prompt": "A different phrasing of the same need, with none of the same keywords."
},
{
"id": "REPLACE-near-miss",
"skill_should_trigger": false,
"note": "Close to this skill's vocabulary but out of its scope: the wrong hardware, an adjacent domain, a question rather than a task.",
"prompt": "Something that sounds like this skill's territory but is not."
},
{
"id": "REPLACE-second-near-miss",
"skill_should_trigger": false,
"prompt": "Another prompt that must not wake this skill up."
}
]
}
SHA-256: 7c23307f42a539df48cbca732247ec4346a3cf26eb2c15b233616e70fe9f36a1