← Files AMDARCHIVED FILE
eval/schema/evals.schema.json
5.75 KB · Sep 30, 2026 · 23:13 UTC
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://github.com/amd/skills/eval/schema/evals.schema.json",
"title": "AMD skill eval dataset",
"description": "One dataset per skill at skills/<name>/evals/evals.json, holding an `evaluations` array. Each evaluation is a prompt plus `skill_should_trigger`, a yes/no answer to whether the skill that owns this file should fire for it. No evaluation names a skill; the folder does that. Routing mode installs the whole catalog and grades the trigger decision for every evaluation. Behavior mode installs only this skill and grades expected_behavior / unexpected_behavior / logs_contain / files_exist, which is why those fields exist only when skill_should_trigger is true. eval/datasets.py is the enforcing implementation and `python eval/run_evals.py --validate` is how you run it; this file is the field reference, and datasets do not link to it.",
"type": "object",
"additionalProperties": false,
"required": ["evaluations"],
"properties": {
"comment": {
"description": "Free-text notes for humans. Ignored by the runner.",
"oneOf": [
{ "type": "string" },
{ "type": "array", "items": { "type": "string" } }
]
},
"evaluations": {
"type": "array",
"minItems": 1,
"description": "Every prompt this skill is graded on, triggering and non-triggering alike. Tier 0 requires at least 3 with skill_should_trigger true and 2 with it false.",
"items": {
"oneOf": [
{ "$ref": "#/$defs/triggeringEvaluation" },
{ "$ref": "#/$defs/nonTriggeringEvaluation" }
]
}
}
},
"$defs": {
"id": {
"type": "string",
"minLength": 1,
"description": "Stable slug, unique across the whole repo because routing pools every skill's evaluations into one run. Used by --only and to name transcript files."
},
"prompt": {
"type": "string",
"minLength": 1,
"description": "What the user says, phrased the way a user would. Do not copy the skill description: that measures string matching rather than routing. A prompt that references a file must either stage it via `workspace` or link a real URL, because the routing workspace holds nothing but the skills tree."
},
"note": {
"type": "string",
"description": "Why this evaluation exists. JSON has no comments, so this is the sanctioned place for one. Ignored by the runner."
},
"triggeringEvaluation": {
"type": "object",
"additionalProperties": false,
"required": ["id", "prompt", "skill_should_trigger"],
"description": "A prompt that must activate the skill that owns this file. Routing grades the trigger decision; behavior mode grades whatever else is asserted here.",
"properties": {
"id": { "$ref": "#/$defs/id" },
"prompt": { "$ref": "#/$defs/prompt" },
"skill_should_trigger": {
"const": true,
"description": "Yes: this skill should fire for this prompt."
},
"expected_behavior": {
"type": "array",
"items": { "type": "string", "minLength": 1 },
"description": "Natural-language things the agent must do, graded by an LLM judge over the transcript and workspace."
},
"unexpected_behavior": {
"type": "array",
"items": { "type": "string", "minLength": 1 },
"description": "Natural-language things the agent must not do, graded by an LLM judge over the transcript and workspace. This is where a skill's guard rails get pinned down."
},
"logs_contain": {
"type": "array",
"items": { "type": "string", "minLength": 1 },
"description": "Case-insensitive substrings that must appear in the run transcript. Deterministic and instant, so prefer these over a judged expectation when the thing you want is a literal (a script name, a flag, a pinned image tag). Do not assert the skill's own name here: routing mode already grades that."
},
"files_exist": {
"type": "array",
"items": { "type": "string", "minLength": 1 },
"description": "Files that must exist when the run ends. Each entry is matched against whole path segments anywhere in the workspace, so `plan.md` is satisfied by `examples/plan.md` and `out/report.md` by `run-1/out/report.md`. Name the artifact, not the directory you hope the agent picks; if the location itself matters, say so in the prompt and grade it as a behavior."
},
"workspace": {
"type": "string",
"minLength": 1,
"description": "Directory relative to the skill root (conventionally evals/files/<name>) whose contents are copied into the agent's workspace before the prompt runs. Use it to hand the agent a starting file to edit instead of describing one in prose."
},
"note": { "$ref": "#/$defs/note" }
}
},
"nonTriggeringEvaluation": {
"type": "object",
"additionalProperties": false,
"required": ["id", "prompt", "skill_should_trigger"],
"description": "A prompt where no skill should fire: the near misses only this skill's owner knows to write, close enough to the domain to be tempting and wrong enough that firing would be a bug. Graded on the trigger decision alone. No skill is ever loaded, so there is no behavior phase, and expected_behavior / unexpected_behavior / logs_contain / files_exist / workspace are rejected here rather than silently ignored.",
"properties": {
"id": { "$ref": "#/$defs/id" },
"prompt": { "$ref": "#/$defs/prompt" },
"skill_should_trigger": {
"const": false,
"description": "No: nothing in the catalog should fire for this prompt."
},
"note": { "$ref": "#/$defs/note" }
}
}
}
}
SHA-256: 30cd9fcf58ba8db6bf729c586f1a568c4ca0263cb8eb0c8af0c288023d1c4125