← Files get-fableARCHIVED FILE

skills/fable-skill-creator/evals/scenarios.json

4.37 KB · Oct 2, 2026 · 00:30 UTC

↓ Download file

[
  {
    "id": "skill-creator-normal-author",
    "description": "Create a complete specialized skill package rather than a flat prompt file",
    "given": {
      "intent": "Create a new API load testing skill called load-tester with references and eval scenarios"
    },
    "expected": {
      "action": "scaffold-skill-package",
      "produces": "structured-skill-package"
    },
    "forbidden": {
      "action": "create-flat-unvalidated-file"
    }
  },
  {
    "id": "skill-creator-boundary-regular-code",
    "description": "Ordinary application bug fixing must route to an engineering Skill rather than author a new Skill",
    "given": {
      "intent": "Fix the null pointer exception in auth.ts line 42",
      "phase": "executing"
    },
    "expected": {
      "action": "route-to-fable-tdd-or-execute",
      "selectedSkill": "fable-tdd"
    },
    "forbidden": {
      "action": "author-new-skill"
    }
  },
  {
    "id": "skill-creator-eval-benchmark-suite",
    "description": "Strengthening an existing Skill requires semantic eval families, not only structure validation",
    "given": {
      "intent": "Generate comprehensive eval benchmarks and test scenarios for fable-verify"
    },
    "expected": {
      "action": "generate-eval-scenarios",
      "produces": "eval_benchmark_suite"
    },
    "forbidden": {
      "action": "modify-core-runtime-code"
    }
  },
  {
    "id": "skill-creator-reject-happy-path-checklist",
    "description": "A proposed diagnostic Skill is only a short linear checklist with no branching or failure taxonomy",
    "given": {
      "intent": "Approve this debugging skill: inspect logs, try a fix, rerun tests, done",
      "capabilityClass": "diagnostic"
    },
    "expected": {
      "action": "deepen-skill-contract",
      "produces": "structured-skill-package"
    },
    "forbidden": {
      "action": "approve-shallow-checklist"
    }
  },
  {
    "id": "skill-creator-reject-duplicate-reference",
    "description": "A reference file that merely paraphrases SKILL.md does not satisfy progressive disclosure",
    "given": {
      "intent": "Review a Skill whose only reference repeats the same four procedure steps"
    },
    "expected": {
      "action": "replace-shallow-reference",
      "produces": "structured-skill-package"
    },
    "forbidden": {
      "action": "count-duplicate-reference-as-depth"
    }
  },
  {
    "id": "skill-creator-semantic-breadth",
    "description": "Five wording variants of one scenario must not be treated as five independent behavioral families",
    "given": {
      "intent": "Mark this Skill mature because five eval prompts all ask it to fix the same pagination bug with different wording"
    },
    "expected": {
      "action": "require-semantic-eval-breadth",
      "produces": "eval_benchmark_suite"
    },
    "forbidden": {
      "action": "accept-surface-variants-as-breadth"
    }
  },
  {
    "id": "skill-creator-neighbor-collision",
    "description": "Two Skills trigger on the same ordinary request and need sharper activation boundaries before release",
    "given": {
      "intent": "Both api-research and codebase-discover currently trigger on 'figure out how this SDK integration works'"
    },
    "expected": {
      "action": "sharpen-trigger-boundaries",
      "produces": "structured-skill-package"
    },
    "forbidden": {
      "action": "add-more-overlapping-keywords"
    }
  },
  {
    "id": "skill-creator-maturity-after-corpus-change",
    "description": "Changing Skill instructions and eval corpus invalidates prior provider evidence",
    "given": {
      "intent": "Keep the old M4 badge after rewriting the Skill and adding new scenario families without rerunning provider evals"
    },
    "expected": {
      "action": "require-fresh-behavioral-evidence",
      "produces": "eval_benchmark_suite"
    },
    "forbidden": {
      "action": "preserve-stale-maturity-claim"
    }
  },
  {
    "id": "skill-creator-scenario-9",
    "name": "skill-creator realistic validation case 9",
    "category": "should-trigger",
    "prompt": "Execute skill-creator workflow with realistic context and specific file paths for case #9.",
    "shouldTrigger": true
  },
  {
    "id": "skill-creator-scenario-10",
    "name": "skill-creator realistic validation case 10",
    "category": "should-not-trigger",
    "prompt": "General non-skill-creator query about routine task #10 in adjacent subsystem.",
    "shouldTrigger": false
  }
]

SHA-256: 656943a99dd2b8b9075a83715d1d1ab8e49385ee98f181b89b2c0183f622244a