← Files AMDARCHIVED FILE

eval/schema/evals.schema.json

5.75 KB · Sep 30, 2026 · 23:13 UTC

↓ Download file

{
  "$schema": "https://json-schema.org/draft/2020-12/schema",
  "$id": "https://github.com/amd/skills/eval/schema/evals.schema.json",
  "title": "AMD skill eval dataset",
  "description": "One dataset per skill at skills/<name>/evals/evals.json, holding an `evaluations` array. Each evaluation is a prompt plus `skill_should_trigger`, a yes/no answer to whether the skill that owns this file should fire for it. No evaluation names a skill; the folder does that. Routing mode installs the whole catalog and grades the trigger decision for every evaluation. Behavior mode installs only this skill and grades expected_behavior / unexpected_behavior / logs_contain / files_exist, which is why those fields exist only when skill_should_trigger is true. eval/datasets.py is the enforcing implementation and `python eval/run_evals.py --validate` is how you run it; this file is the field reference, and datasets do not link to it.",
  "type": "object",
  "additionalProperties": false,
  "required": ["evaluations"],
  "properties": {
    "comment": {
      "description": "Free-text notes for humans. Ignored by the runner.",
      "oneOf": [
        { "type": "string" },
        { "type": "array", "items": { "type": "string" } }
      ]
    },
    "evaluations": {
      "type": "array",
      "minItems": 1,
      "description": "Every prompt this skill is graded on, triggering and non-triggering alike. Tier 0 requires at least 3 with skill_should_trigger true and 2 with it false.",
      "items": {
        "oneOf": [
          { "$ref": "#/$defs/triggeringEvaluation" },
          { "$ref": "#/$defs/nonTriggeringEvaluation" }
        ]
      }
    }
  },
  "$defs": {
    "id": {
      "type": "string",
      "minLength": 1,
      "description": "Stable slug, unique across the whole repo because routing pools every skill's evaluations into one run. Used by --only and to name transcript files."
    },
    "prompt": {
      "type": "string",
      "minLength": 1,
      "description": "What the user says, phrased the way a user would. Do not copy the skill description: that measures string matching rather than routing. A prompt that references a file must either stage it via `workspace` or link a real URL, because the routing workspace holds nothing but the skills tree."
    },
    "note": {
      "type": "string",
      "description": "Why this evaluation exists. JSON has no comments, so this is the sanctioned place for one. Ignored by the runner."
    },
    "triggeringEvaluation": {
      "type": "object",
      "additionalProperties": false,
      "required": ["id", "prompt", "skill_should_trigger"],
      "description": "A prompt that must activate the skill that owns this file. Routing grades the trigger decision; behavior mode grades whatever else is asserted here.",
      "properties": {
        "id": { "$ref": "#/$defs/id" },
        "prompt": { "$ref": "#/$defs/prompt" },
        "skill_should_trigger": {
          "const": true,
          "description": "Yes: this skill should fire for this prompt."
        },
        "expected_behavior": {
          "type": "array",
          "items": { "type": "string", "minLength": 1 },
          "description": "Natural-language things the agent must do, graded by an LLM judge over the transcript and workspace."
        },
        "unexpected_behavior": {
          "type": "array",
          "items": { "type": "string", "minLength": 1 },
          "description": "Natural-language things the agent must not do, graded by an LLM judge over the transcript and workspace. This is where a skill's guard rails get pinned down."
        },
        "logs_contain": {
          "type": "array",
          "items": { "type": "string", "minLength": 1 },
          "description": "Case-insensitive substrings that must appear in the run transcript. Deterministic and instant, so prefer these over a judged expectation when the thing you want is a literal (a script name, a flag, a pinned image tag). Do not assert the skill's own name here: routing mode already grades that."
        },
        "files_exist": {
          "type": "array",
          "items": { "type": "string", "minLength": 1 },
          "description": "Files that must exist when the run ends. Each entry is matched against whole path segments anywhere in the workspace, so `plan.md` is satisfied by `examples/plan.md` and `out/report.md` by `run-1/out/report.md`. Name the artifact, not the directory you hope the agent picks; if the location itself matters, say so in the prompt and grade it as a behavior."
        },
        "workspace": {
          "type": "string",
          "minLength": 1,
          "description": "Directory relative to the skill root (conventionally evals/files/<name>) whose contents are copied into the agent's workspace before the prompt runs. Use it to hand the agent a starting file to edit instead of describing one in prose."
        },
        "note": { "$ref": "#/$defs/note" }
      }
    },
    "nonTriggeringEvaluation": {
      "type": "object",
      "additionalProperties": false,
      "required": ["id", "prompt", "skill_should_trigger"],
      "description": "A prompt where no skill should fire: the near misses only this skill's owner knows to write, close enough to the domain to be tempting and wrong enough that firing would be a bug. Graded on the trigger decision alone. No skill is ever loaded, so there is no behavior phase, and expected_behavior / unexpected_behavior / logs_contain / files_exist / workspace are rejected here rather than silently ignored.",
      "properties": {
        "id": { "$ref": "#/$defs/id" },
        "prompt": { "$ref": "#/$defs/prompt" },
        "skill_should_trigger": {
          "const": false,
          "description": "No: nothing in the catalog should fire for this prompt."
        },
        "note": { "$ref": "#/$defs/note" }
      }
    }
  }
}

SHA-256: 30cd9fcf58ba8db6bf729c586f1a568c4ca0263cb8eb0c8af0c288023d1c4125