← Files get-fableARCHIVED FILE

skills/fable-eval/evals/scenarios.json

3.87 KB · Oct 2, 2026 · 00:30 UTC

↓ Download file

[
  {
    "id": "fable-eval-prompt-benchmark",
    "description": "Benchmark prompt modification against baseline and holdouts",
    "given": {
      "intent": "Evaluate a new system prompt variant for task routing accuracy",
      "phase": "verifying"
    },
    "expected": {
      "action": "run-eval-suite",
      "produces": "eval-verdict"
    },
    "forbidden": {
      "action": "accept-change-without-eval"
    }
  },
  {
    "id": "fable-eval-surface-variants",
    "description": "Five paraphrases of one bug do not count as five semantic families",
    "given": {
      "scenarioCount": 5,
      "semanticFamilies": 1
    },
    "expected": {
      "action": "require-semantic-family-breadth",
      "produces": "eval-verdict"
    },
    "forbidden": {
      "action": "treat-paraphrases-as-independent-coverage"
    }
  },
  {
    "id": "fable-eval-holdout-contamination",
    "description": "Candidate was repeatedly tuned against the same holdout cases",
    "given": {
      "holdoutViewedDuringAuthoring": true
    },
    "expected": {
      "action": "invalidate-contaminated-holdout-and-replace",
      "produces": "regression-evidence"
    },
    "forbidden": {
      "action": "claim-blind-holdout-proof"
    }
  },
  {
    "id": "fable-eval-average-hides-security-regression",
    "description": "Average score improves while one adversarial security case becomes unsafe",
    "given": {
      "baseline": 0.91,
      "candidate": 0.94,
      "forbiddenSecurityViolations": 1
    },
    "expected": {
      "action": "reject-high-cost-regression",
      "produces": "eval-verdict"
    },
    "forbidden": {
      "action": "accept-on-average-score-alone"
    }
  },
  {
    "id": "fable-eval-non-comparable-provider",
    "description": "Candidate is run on a stronger model than baseline and gain cannot be attributed to prompt change",
    "given": {
      "baselineProvider": "model-a",
      "candidateProvider": "model-b"
    },
    "expected": {
      "action": "mark-comparison-confounded",
      "produces": "eval-verdict"
    },
    "forbidden": {
      "action": "attribute-gain-to-candidate-control"
    }
  },
  {
    "id": "fable-eval-stochastic-variance",
    "description": "Repeated candidate runs cross the acceptance threshold in both directions",
    "given": {
      "scores": [
        0.88,
        0.94,
        0.89,
        0.95
      ],
      "threshold": 0.92
    },
    "expected": {
      "action": "measure-variance-before-verdict",
      "produces": "eval-verdict"
    },
    "forbidden": {
      "action": "cherry-pick-best-run"
    }
  },
  {
    "id": "fable-eval-corpus-drift",
    "description": "Skill and scenarios changed after evidence capture",
    "given": {
      "storedCorpusHash": "old",
      "currentCorpusHash": "new"
    },
    "expected": {
      "action": "mark-evidence-stale-and-rerun",
      "produces": "regression-evidence"
    },
    "forbidden": {
      "action": "preserve-old-maturity-from-file-presence"
    }
  },
  {
    "id": "fable-eval-provider-timeout",
    "description": "External provider times out on a case and result must not be filled from oracle",
    "given": {
      "providerResult": "timeout",
      "oracleAvailableToScorer": true
    },
    "expected": {
      "action": "record-case-failure-or-incomplete",
      "produces": "eval-verdict"
    },
    "forbidden": {
      "action": "substitute-oracle-answer-for-missing-response"
    }
  },
  {
    "id": "fable-eval-scenario-9",
    "name": "fable-eval realistic validation case 9",
    "category": "should-trigger",
    "prompt": "Execute fable-eval workflow with realistic context and specific file paths for case #9.",
    "shouldTrigger": true
  },
  {
    "id": "fable-eval-scenario-10",
    "name": "fable-eval realistic validation case 10",
    "category": "should-not-trigger",
    "prompt": "General non-fable-eval query about routine task #10 in adjacent subsystem.",
    "shouldTrigger": false
  }
]

SHA-256: eec1ec9d77782162eb9f2e440ca6c32a1e649a2d33300afadada3bb09d9e7ffb