← Files Biohub ESMARCHIVED FILE

scripts/biohub_esm_lib/starter_examples.py

68.8 KB · Sep 30, 2026 · 23:14 UTC

↓ Download file

"""Deterministic validation for packaged Biohub ESM starter examples."""

from __future__ import annotations

import hashlib
import json
import re
from pathlib import Path
from typing import Any

from .activation import select_skill
from .errors import ValidationError
from .routing import RouteRequest, route_request
from .validation import (
    validate_atlas_search_sequence,
    validate_esmc_sequence,
    validate_fold_config,
    validate_fold_input,
)

CANONICAL_AMINO_ACIDS = "ACDEFGHIKLMNPQRSTVWY"
EXPECTED_SEQUENCE = {
    "id": "gb1-1pga-chain-a",
    "name": "Streptococcal protein G B1 domain (GB1), RCSB PDB 1PGA chain A",
    "literal": "MTYKLILNGKTLKGETTTEAVDAATAEKVFKQYANDNGVDGEWTYDDATKTFTVTE",
    "source": {
        "title": "RCSB PDB 1PGA: B1 immunoglobulin-binding domain of streptococcal protein G",
        "reference": "https://www.rcsb.org/structure/1PGA",
        "fasta_url": "https://www.rcsb.org/fasta/entry/1PGA/display",
        "accession": "1PGA",
        "chain": "A",
        "entry_revision": "1.4",
        "snapshot_date": "2026-07-12",
        "raw_fasta_bytes": 115,
        "raw_fasta_sha256": ("a56ce69e9e80a566367b6b4b3a01a263d61e9b5c58dd55e9eb8e49a50bd6b561"),
        "usage": "Exact 56-residue chain A sequence from the experimentally determined 1PGA entry",
        "numbering": (
            "One-based residue positions on the exact 56-residue literal sequence in this contract"
        ),
    },
    "length": 56,
    "sha256": "7e859d82171047700fd3e9632f7a47eab4a39baedc8c3316d2fc62d3ce2260bb",
}
EXPECTED_PROMPTS = {
    "gb1-esmc-w43f-masked-llr": "What might W43F do to GB1?",
    "gb1-esmfold2-fast-fold": "Show me what GB1 looks like.",
    "gb1-atlas-similarity-search": "Find proteins similar to GB1.",
}
EXPECTED_ROUTES = {
    "gb1-esmc-w43f-masked-llr": {
        "skill": "esmc",
        "model": "esmc-600m-2024-12",
        "provider": "biohub-managed",
        "may_incur_cost": True,
        "confirmation_boundary": "not_required_for_managed_tutorial_scale",
        "prompt_authorizes_execution": True,
        "required_confirmation": (
            "not required; report the credits consumed with the result instead"
        ),
        "request_policy": ("exactly one managed logits request with no implicit retries"),
        "local_alternative": ("biohub/ESMC-600M at the pinned revision on user-owned compute"),
    },
    "gb1-esmfold2-fast-fold": {
        "skill": "esmfold2",
        "model": "esmfold2-fast-2026-05",
        "provider": "biohub-managed",
        "may_incur_cost": True,
        "confirmation_boundary": "not_required_for_managed_tutorial_scale",
        "prompt_authorizes_execution": True,
        "required_confirmation": (
            "not required; report the credits consumed with the result instead"
        ),
        "request_policy": "exactly one managed fold request with no implicit retries",
        "local_alternative": ("biohub/ESMFold2-Fast at the pinned revision on user-owned compute"),
    },
    "gb1-atlas-similarity-search": {
        "skill": "esm-atlas",
        "model": "not-applicable-public-data-api",
        "provider": "esm-atlas-v1alpha1-public-api",
        "may_incur_cost": False,
        "confirmation_boundary": "not_required_for_public_alpha_read",
        "prompt_authorizes_execution": True,
        "required_confirmation": "not required for the documented public alpha read",
        "request_policy": ("exactly one public search request with zero detail follow-ups"),
        "local_alternative": ("anonymous ESM Atlas S3 data for deliberately scoped bulk analysis"),
    },
}
EXPECTED_MANAGED_WORKFLOWS = {
    "gb1-esmc-w43f-masked-llr": (
        "Run status-only preflight first; if managed access is missing, give the user the key page from $biohub-esm-setup and resume automatically once it is configured.",
        "Resolve the natural launcher to the pinned GB1 fixture and validate its sequence, SHA-256, length, and mutation numbering without making a provider call.",
        "Validate the managed model ID, endpoint, and pinned client SDK revision, tokenize locally with the pinned SDK, and make exactly one managed logits request without implicit retries.",
        "Compute the W43F masked log-likelihood ratio with one documented masking and scoring method.",
        "Serialize the raw response, score, deterministic SVG score card, parameters, and provenance without claiming fitness, stability, or function, then display the SVG inline and note the credits consumed.",
    ),
    "gb1-esmfold2-fast-fold": (
        "Using the same resolved <python-3.12-with-pinned-esm> interpreter that will execute the request, run status-only preflight --endpoint fold first; if managed access is missing, give the user the key page from $biohub-esm-setup and resume automatically once it is configured.",
        "Resolve the natural launcher to the pinned GB1 fixture and validate its sequence, SHA-256, length, and Fast single-sequence constraints without making a provider call.",
        "Run the pinned managed-post --endpoint fold command exactly once, without implicit retries, from examples/gb1-esmfold2-fast-fold-request.json with include_pae=true and the validated Fast parameters.",
        "Write the PDB, retained presentation request, normalized confidence result, raw response, provenance, and cryptographic artifact hashes; after successful creation, consume presentation-request.json, open its verified absolute prediction.pdb path once with its exact retained openIntentId, retain and verify the same-task session, request predicted-confidence coloring only when ready, report pending or unavailable presentation separately, and note the credits consumed.",
    ),
}
EXPECTED_EXAMPLES = {example_id: route["skill"] for example_id, route in EXPECTED_ROUTES.items()}
EXPECTED_PRODUCTION_ROUTES = {
    "gb1-esmc-w43f-masked-llr": (
        RouteRequest(task="esmc"),
        "biohub",
        "esmc-600m-2024-12",
        "ESM_API_KEY",
    ),
    "gb1-esmfold2-fast-fold": (
        RouteRequest(task="fold"),
        "biohub",
        "esmfold2-fast-2026-05",
        "ESM_API_KEY",
    ),
    "gb1-atlas-similarity-search": (
        RouteRequest(task="atlas"),
        "atlas-api",
        None,
        None,
    ),
}
EXPECTED_PROVIDER_BY_PRODUCTION_ROUTE = {
    "biohub": "biohub-managed",
    "atlas-api": "esm-atlas-v1alpha1-public-api",
}
EXPECTED_ARTIFACTS = {
    "gb1-esmc-w43f-masked-llr": (
        {"name": "raw-response.json", "format": "JSON", "required": True},
        {
            "name": "mutation-score.json",
            "format": "JSON",
            "required": True,
            "derivation": (
                "Derive the documented masked log-likelihood ratio from the raw response, "
                "preserve both log probabilities and one-based numbering, checksum it, "
                "and include it in provenance."
            ),
        },
        {
            "name": "mutation-score.svg",
            "format": "SVG",
            "required": True,
            "derivation": (
                "Render a deterministic self-contained score card from mutation-score.json "
                "and label it as a model hypothesis rather than fitness, stability, or function."
            ),
        },
        {"name": "provenance.json", "format": "JSON", "required": True},
    ),
    "gb1-esmfold2-fast-fold": (
        {"name": "prediction.pdb", "format": "PDB", "required": True},
        {"name": "presentation-request.json", "format": "JSON", "required": True},
        {"name": "result.json", "format": "JSON", "required": True},
        {"name": "raw-response.json", "format": "JSON", "required": True},
        {"name": "provenance.json", "format": "JSON", "required": True},
    ),
    "gb1-atlas-similarity-search": (
        {"name": "raw-response.json", "format": "JSON", "required": True},
        {"name": "result.json", "format": "JSON", "required": True},
        {"name": "provenance.json", "format": "JSON", "required": True},
    ),
}
EXPECTED_ESMC_EXECUTION = {
    "request_count": 1,
    "method": "POST",
    "endpoint": "https://biohub.ai/api/v1/logits",
    "sequence": EXPECTED_SEQUENCE["literal"],
    "mutation": "W43F",
    "model": "esmc-600m-2024-12",
    "scoring_method": "masked-log-likelihood-ratio",
    "command": [
        "<python-3.12-with-pinned-esm>",
        "<plugin-root>/scripts/biohub_esm.py",
        "esmc-mutation-score",
        "--sequence",
        EXPECTED_SEQUENCE["literal"],
        "--mutation",
        "W43F",
        "--model",
        "esmc-600m-2024-12",
        "--output-dir",
        "/absolute/path/gb1-esmc-w43f-masked-llr",
    ],
    "output_dir": "/absolute/path/gb1-esmc-w43f-masked-llr",
    "artifacts": [
        "raw-response.json",
        "mutation-score.json",
        "mutation-score.svg",
        "provenance.json",
    ],
    "follow_up_requests": [],
}
EXPECTED_ESMC_QUALIFICATION_RESPONSE = {
    "format": "exact-lines-v1",
    "lines": [
        (
            "ESMC input: GB1 1PGA chain A; 56 residues; mutation W43F in one-based residue "
            "numbering; sequence SHA-256 "
            "7e859d82171047700fd3e9632f7a47eab4a39baedc8c3316d2fc62d3ce2260bb."
        ),
        (
            "ESMC request: exactly one Biohub managed POST /api/v1/logits with model "
            "esmc-600m-2024-12 using ESM_API_KEY; no implicit retries."
        ),
        (
            "Mutation score formula: use one masked context and compute the alternate logit "
            "minus the wild-type logit."
        ),
        (
            "ESMC artifacts: raw-response.json, mutation-score.json, mutation-score.svg, "
            "and provenance.json with checksums."
        ),
        (
            "ESMC provenance: preserve the route, provider, endpoint, exact model ID, any "
            "managed model revision exposed, pinned ESM and Transformers SDK revisions, input "
            "digest, parameters, timestamps, raw-response and artifact checksums, masked "
            "context, both natural-log probabilities, one-based residue index, zero-based "
            "tensor index, and W43F score."
        ),
        (
            "ESMC cost: running this consumes a small number of provider credits, reported "
            "with the result. A managed request at this scale runs without a separate "
            "confirmation; Modal, self-hosted, and bulk transfers still ask first."
        ),
        (
            "Scientific status: model output is a model hypothesis, not an experimental truth, "
            "and experimental validation is required before biological interpretation."
        ),
        "No provider call or request was sent.",
    ],
}
EXPECTED_FOLD_EXECUTION = {
    "request_count": 1,
    "method": "POST",
    "endpoint": "https://biohub.ai/api/v1/fold",
    "request_file": "examples/gb1-esmfold2-fast-fold-request.json",
    "request": {
        "model": "esmfold2-fast-2026-05",
        "sequence": EXPECTED_SEQUENCE["literal"],
        "msa": None,
        "include_distogram": False,
        "include_pae": True,
        "include_pair_chains_iptm": False,
        "num_sampling_steps": 100,
        "num_loops": 20,
        "lm_dropout": 0.3,
        "lm_mask_pct": 0.1,
        "msa_max_depth": 1024,
        "msa_column_mask_rate": 0.1,
        "include_embeddings": False,
    },
    "command": [
        "<python-3.12-with-pinned-esm>",
        "<plugin-root>/scripts/biohub_esm.py",
        "managed-post",
        "--endpoint",
        "fold",
        "--input",
        "<plugin-root>/examples/gb1-esmfold2-fast-fold-request.json",
        "--output-dir",
        "/absolute/path/gb1-esmfold2-fast-fold",
    ],
    "output_dir": "/absolute/path/gb1-esmfold2-fast-fold",
    "offline_validation": {
        "function": "validate_fold_config",
        "model": "esmfold2-fast-2026-05",
        "endpoint": "fold",
    },
    "follow_up_requests": [],
}
EXPECTED_FOLD_QUALIFICATION_RESPONSE = {
    "format": "exact-lines-v1",
    "lines": [
        (
            "Fold input: GB1 1PGA chain A; 56 residues; sequence SHA-256 "
            "7e859d82171047700fd3e9632f7a47eab4a39baedc8c3316d2fc62d3ce2260bb; msa=null."
        ),
        (
            "Fold request: exactly one Biohub managed POST /api/v1/fold with model "
            "esmfold2-fast-2026-05 using examples/gb1-esmfold2-fast-fold-request.json; "
            "no implicit retries."
        ),
        (
            "Fold parameters: include_distogram=false; include_embeddings=false; "
            "include_pair_chains_iptm=false; num_sampling_steps=100; num_loops=20; "
            "lm_dropout=0.3; lm_mask_pct=0.1; msa_max_depth=1024; "
            "msa_column_mask_rate=0.1."
        ),
        "Fold pAE request parameter: include_pae=true.",
        (
            "Fold confidence scope: provider-native pLDDT=per-residue on the 0-1 scale; "
            "pAE=residue-pair in angstroms."
        ),
        (
            "Fold artifacts: prediction.pdb, presentation-request.json, result.json, "
            "raw-response.json, and provenance.json with checksums."
        ),
        (
            "Fold presentation: after successful artifact creation, consume the validated "
            "presentation-request.json and pass its verified absolute prediction.pdb path "
            "once to the available molecular-structure-viewing capability using its exact "
            "retained openIntentId. Generate a new stable ID only for a legacy artifact set "
            "without presentation-request.json. Retain the returned session within this task, "
            "verify the primary object, and request predicted-confidence coloring only when "
            "ready, labeling the coordinate B-factor field as serializer-scaled 0-100 pLDDT "
            "rather than an experimental temperature factor. If presentation is pending or "
            "unavailable, report that separately and return the checksummed artifacts without "
            "invalidating the fold."
        ),
        (
            "Fold cost: running this consumes a small number of provider credits, reported "
            "with the result. A managed request at this scale runs without a separate "
            "confirmation; Modal, self-hosted, and bulk transfers still ask first."
        ),
        (
            "Scientific status: model output is a static conformational model hypothesis, "
            "not an experimental truth, and experimental validation is required before "
            "biological interpretation."
        ),
        "No provider call or request was sent.",
    ],
}
EXPECTED_ATLAS_QUALIFICATION_RESPONSE = {
    "format": "exact-lines-v1",
    "lines": [
        (
            "Atlas authentication: ESM Atlas currently does not require a client-side "
            "API key or client authentication."
        ),
        "Atlas similarity-search ceiling: at most 800 residues.",
        "Atlas on-demand-fold ceiling: at most 699 residues.",
        "Atlas starter fold scope: no on-demand fold is planned or authorized.",
        (
            "Atlas starter search (public v1alpha1): 56-residue GB1 query; "
            "topk_results=10 (up to 10 actual ranked hits); topk_features=20; "
            "min_similarity=0.5; include_cluster_info=true."
        ),
        (
            "Atlas execution plan: exactly one similarity search; zero protein or "
            "cluster-detail follow-ups."
        ),
        (
            "Atlas presentation: use only coordinates embedded in the one search response; "
            "if absent and the result is non-empty, make at most one public pLDDT thumbnail "
            "request for the top-ranked hit."
        ),
        (
            "Atlas structure status: absent authoritative origin metadata, returned "
            "coordinates are model hypotheses and pLDDT thumbnails are model-confidence "
            "summaries, not experimental structures or evidence; experimental validation "
            "is required."
        ),
        (
            "Atlas evidence: preserve raw-response.json, result.json, and provenance.json "
            "with endpoint and request parameters, access timestamps, response and artifact "
            "checksums, v1alpha1 schema, and CC-BY-4.0 attribution."
        ),
        "Atlas schema: v1alpha1 is alpha and unstable.",
        "No Atlas request was sent and no results were obtained.",
    ],
}
EXPECTED_ATLAS_EXECUTION = {
    "request_count": 1,
    "method": "GET",
    "endpoint": "https://biohub.ai/esm/protein/api/v1alpha1/similarity-search",
    "query": {
        "sequence": EXPECTED_SEQUENCE["literal"],
        "topk_results": 10,
        "topk_features": 20,
        "min_similarity": 0.5,
        "cluster_pct_characterized_max": None,
        "include_cluster_info": True,
    },
    "command": [
        "python3",
        "<plugin-root>/scripts/biohub_esm.py",
        "atlas",
        "search",
        "--sequence",
        EXPECTED_SEQUENCE["literal"],
        "--topk-results",
        "10",
        "--topk-features",
        "20",
        "--min-similarity",
        "0.5",
        "--include-cluster-info",
        "--output-dir",
        "/absolute/path/gb1-atlas-similarity-search",
    ],
    "output_dir": "/absolute/path/gb1-atlas-similarity-search",
    "response_contract": {
        "maximum_hits": 10,
        "required_hit_fields": [
            "protein_hash",
            "protein_accession",
            "sequence_length",
            "similarity_score",
        ],
        "unique_by": "protein_hash",
        "similarity_range": [0.0, 1.0],
        "minimum_similarity": 0.5,
        "order": "nonincreasing-similarity",
    },
    "detail_follow_up_requests": [],
    "presentation_request": {
        "condition": "non-empty search result with no returned coordinate artifact",
        "maximum_request_count": 1,
        "method": "GET",
        "endpoint_template": (
            "https://biohub.ai/esm/protein/api/v1alpha1/proteins/<top-hit-md5>/thumbnail/plddt"
        ),
        "command": [
            "python3",
            "<plugin-root>/scripts/biohub_esm.py",
            "atlas",
            "thumbnail",
            "--protein-hash",
            "<top-hit-md5>",
            "--thumbnail-type",
            "plddt",
            "--output",
            "/absolute/path/gb1-atlas-similarity-search/top-hit-plddt.png",
        ],
        "artifact": "top-hit-plddt.png",
        "provenance_artifact": "top-hit-plddt.png.provenance.json",
    },
    "viewer_selection": (
        "first-ranked normalized hit containing pdb_artifact; otherwise use the conditional "
        "pLDDT thumbnail presentation request for a non-empty result"
    ),
}
EXPECTED_ATLAS_OPTIONAL_ARTIFACTS = [
    {
        "name_pattern": "search-<n>.pdb",
        "format": "PDB",
        "condition": "Only when the provider embeds coordinates in a returned search hit.",
    },
    {
        "name": "top-hit-plddt.png",
        "format": "PNG",
        "condition": (
            "Only for a non-empty search result when no returned hit contains a coordinate "
            "artifact."
        ),
    },
    {
        "name": "top-hit-plddt.png.provenance.json",
        "format": "JSON",
        "condition": "Written with the conditional top-hit pLDDT thumbnail.",
    },
]

EXPECTED_STRUCTURE_VIEWER_CONTRACT = {
    "capability": "interactive-molecular-structure-viewing",
    "request_artifact": "presentation-request.json",
    "artifact_path": "verified-absolute",
    "open_intent_field": "openIntentId",
    "open_intent_policy": "reuse-exact-emitted-value",
    "open_intent_reuse": "delivery-retry-only",
    "legacy_open_intent_policy": "generate-new-stable-only-when-request-artifact-missing",
    "open_policy": "once",
    "session_policy": "retain-returned-session-within-same-task",
    "verification": "list-primary-object",
    "style_when": "ready",
    "predicted_confidence_style": {
        "semantic": "predicted-confidence",
        "source": "coordinate-b-factor",
    },
    "timed_out_mutation": "inspect-acknowledged-state-before-retry",
    "blocked_outcomes": ["pending", "unavailable"],
    "scientific_success_independent": True,
}

EXPECTED_PRESENTATIONS = {
    "gb1-esmc-w43f-masked-llr": {
        "mode": "inline-image",
        "automatic": True,
        "primary_artifact": "mutation-score.svg",
        "capability": "inline-image-display",
        "style": (
            "Display the deterministic score card inline and label it as a model score rather "
            "than a biological effect."
        ),
        "fallback": (
            "Return the verified SVG and JSON artifact paths if inline image display is "
            "unavailable; the JSON artifacts remain authoritative."
        ),
    },
    "gb1-esmfold2-fast-fold": {
        "mode": "structure-viewer",
        "automatic": True,
        "primary_artifact": "prediction.pdb",
        "capability": "interactive-molecular-structure-viewing",
        "contract": "structure_viewer_contract",
        "style": (
            "Request predicted-confidence coloring only when the viewer is ready and label "
            "coordinate B-factor values as pLDDT rather than experimental temperature factors."
        ),
        "fallback": (
            "If presentation is pending, retain the intent and report it as pending; if the "
            "capability is unavailable, return the verified PDB and result/provenance paths. "
            "Neither presentation outcome invalidates the fold."
        ),
    },
    "gb1-atlas-similarity-search": {
        "mode": "structure-or-inline-image",
        "automatic": True,
        "primary_artifact": (
            "the first-ranked normalized hit containing pdb_artifact; otherwise "
            "top-hit-plddt.png for a non-empty result"
        ),
        "capability": "interactive-molecular-structure-viewing-or-inline-image-display",
        "style": (
            "Open returned coordinates through the structure-viewing capability, or display one "
            "pLDDT thumbnail inline when coordinates are absent."
        ),
        "fallback": (
            "For a pending viewer retain the intent and report it as pending; for an empty result "
            "or unavailable presentation return verified JSON artifacts. Never trigger an "
            "on-demand fold implicitly."
        ),
    },
}
PLACEHOLDER_PATTERN = re.compile(
    r"(?:\bthis protein\b|\bthis sequence\b|\byour sequence\b|"
    r"\bthis complex\b|\bthese sequences\b|\bprotein sequence here\b|<sequence>|\.{3})",
    re.IGNORECASE,
)
MUTATION_PATTERN = re.compile(r"\b([A-Z])([1-9][0-9]*)([A-Z])\b")
MARKETPLACE_SEQUENCE_LITERAL_PATTERN = re.compile(
    r"\b[ACDEFGHIKLMNPQRSTVWY]{30,}\b",
    re.IGNORECASE,
)
MARKETPLACE_INVOCATION_PATTERN = re.compile(r"(?<![A-Za-z0-9_])@[A-Za-z0-9][A-Za-z0-9_-]*\b")


class StarterExampleError(ValueError):
    """Raised when the starter-example contract is incomplete or inconsistent."""


def _object(value: Any, label: str) -> dict[str, Any]:
    if not isinstance(value, dict):
        raise StarterExampleError(f"{label} must be an object")
    return value



def _validate_structure_viewer_contract(contract: dict[str, Any]) -> None:
    viewer_contract = _object(
        contract.get("structure_viewer_contract"), "structure_viewer_contract"
    )
    if viewer_contract != EXPECTED_STRUCTURE_VIEWER_CONTRACT:
        raise StarterExampleError("structure_viewer_contract has drifted")


def _nonempty_string(value: Any, label: str) -> str:
    if not isinstance(value, str) or not value.strip():
        raise StarterExampleError(f"{label} must be a non-empty string")
    return value


def _string_list(value: Any, label: str, *, minimum: int = 1) -> list[str]:
    if not isinstance(value, list) or len(value) < minimum:
        raise StarterExampleError(f"{label} must contain at least {minimum} item(s)")
    for index, item in enumerate(value):
        _nonempty_string(item, f"{label}[{index}]")
    return value


def _load_json_object(path: Path, label: str) -> dict[str, Any]:
    try:
        decoded = json.loads(path.read_text(encoding="utf-8"))
    except (OSError, json.JSONDecodeError) as exc:
        raise StarterExampleError(f"cannot load {label}: {exc}") from exc
    return _object(decoded, label)


def load_starter_examples(path: Path) -> dict[str, Any]:
    return _load_json_object(path, "starter example contract")


def load_tutorial_use_cases(path: Path) -> dict[str, Any]:
    return _load_json_object(path, "tutorial use-case contract")


def _validate_sequence(contract: dict[str, Any]) -> tuple[dict[str, Any], str]:
    sequence = _object(contract.get("sequence"), "sequence")
    for field in ("id", "name", "literal", "sha256"):
        _nonempty_string(sequence.get(field), f"sequence.{field}")
    literal = sequence["literal"]
    if literal != literal.upper() or not literal.isalpha():
        raise StarterExampleError("sequence.literal must be uppercase amino-acid letters")
    invalid = sorted(set(literal) - set(CANONICAL_AMINO_ACIDS))
    if invalid:
        raise StarterExampleError(
            f"sequence.literal contains non-canonical residue(s): {''.join(invalid)}"
        )
    if sequence.get("length") != len(literal):
        raise StarterExampleError("sequence.length does not match sequence.literal")
    digest = hashlib.sha256(literal.encode("ascii")).hexdigest()
    if sequence.get("sha256") != digest:
        raise StarterExampleError("sequence.sha256 does not match sequence.literal")

    source = _object(sequence.get("source"), "sequence.source")
    for field in (
        "title",
        "reference",
        "fasta_url",
        "accession",
        "chain",
        "entry_revision",
        "snapshot_date",
        "raw_fasta_sha256",
        "usage",
        "numbering",
    ):
        _nonempty_string(source.get(field), f"sequence.source.{field}")
    for field in ("reference", "fasta_url"):
        if not source[field].startswith("https://"):
            raise StarterExampleError(f"sequence.source.{field} must be an HTTPS URL")
    if re.fullmatch(r"[0-9]{4}-[0-9]{2}-[0-9]{2}", source["snapshot_date"]) is None:
        raise StarterExampleError("sequence.source.snapshot_date must be YYYY-MM-DD")
    if source["accession"] != "1PGA" or source["chain"] != "A":
        raise StarterExampleError("sequence source must remain RCSB 1PGA chain A")
    if source["entry_revision"] != "1.4":
        raise StarterExampleError("sequence source entry_revision has drifted")
    if source.get("raw_fasta_bytes") != 115:
        raise StarterExampleError("sequence source raw_fasta_bytes has drifted")
    if source["raw_fasta_sha256"] != (
        "a56ce69e9e80a566367b6b4b3a01a263d61e9b5c58dd55e9eb8e49a50bd6b561"
    ):
        raise StarterExampleError("sequence source raw_fasta_sha256 has drifted")
    if sequence != EXPECTED_SEQUENCE:
        raise StarterExampleError(
            "sequence and source metadata must exactly match the pinned RCSB 1PGA chain A snapshot"
        )
    return sequence, literal


def _validate_mutations(
    example: dict[str, Any], sequence: str, input_validation: dict[str, Any]
) -> None:
    mutations = input_validation.get("mutations")
    if not isinstance(mutations, list):
        raise StarterExampleError(f"{example['id']}.input_validation.mutations must be a list")
    prompt_mutations = {"".join(parts) for parts in MUTATION_PATTERN.findall(example["prompt"])}
    if set(mutations) != prompt_mutations:
        raise StarterExampleError(
            f"{example['id']} mutations must exactly match mutations in its prompt"
        )
    for mutation in mutations:
        if not isinstance(mutation, str):
            raise StarterExampleError(f"{example['id']} mutation must be a string")
        match = MUTATION_PATTERN.fullmatch(mutation)
        if match is None:
            raise StarterExampleError(f"{example['id']} has invalid mutation {mutation!r}")
        wild_type, position_text, alternate = match.groups()
        position = int(position_text)
        if not 1 <= position <= len(sequence):
            raise StarterExampleError(
                f"{example['id']} mutation {mutation} is outside the literal sequence"
            )
        if sequence[position - 1] != wild_type:
            raise StarterExampleError(
                f"{example['id']} mutation {mutation} has the wrong wild-type residue"
            )
        if alternate not in CANONICAL_AMINO_ACIDS or alternate == wild_type:
            raise StarterExampleError(
                f"{example['id']} mutation {mutation} has an invalid alternate residue"
            )
    mutant = list(sequence)
    expected_differences: set[int] = set()
    for mutation in mutations:
        match = MUTATION_PATTERN.fullmatch(mutation)
        assert match is not None
        _, position_text, alternate = match.groups()
        position = int(position_text)
        mutant[position - 1] = alternate
        expected_differences.add(position)
    observed_differences = {
        index + 1
        for index, (wild_type, alternate) in enumerate(zip(sequence, mutant, strict=True))
        if wild_type != alternate
    }
    if len(mutant) != len(sequence) or observed_differences != expected_differences:
        raise StarterExampleError(
            f"{example['id']} mutation application does not preserve exact one-based identity"
        )


def _validate_route(example: dict[str, Any]) -> None:
    route = _object(example.get("route"), f"{example['id']}.route")
    for field in (
        "skill",
        "model",
        "provider",
        "confirmation_boundary",
        "required_confirmation",
        "request_policy",
        "local_alternative",
    ):
        _nonempty_string(route.get(field), f"{example['id']}.route.{field}")
    for field in ("may_incur_cost", "prompt_authorizes_execution"):
        if not isinstance(route.get(field), bool):
            raise StarterExampleError(f"{example['id']}.route.{field} must be boolean")
    if route["skill"] != EXPECTED_EXAMPLES[example["id"]]:
        raise StarterExampleError(f"{example['id']} is assigned to the wrong skill")
    # Confirmation is a function of the route, not of whether money moves at all.
    # Managed tutorial-scale calls cost fractions of a cent and the user already
    # chose the prompt, so they run; Modal, self-hosted, and bulk transfers confirm.
    if route["provider"] == "biohub-managed":
        if route["confirmation_boundary"] != "not_required_for_managed_tutorial_scale":
            raise StarterExampleError(
                f"{example['id']} managed starter must state its exact confirmation boundary"
            )
        if route["prompt_authorizes_execution"] is not True:
            raise StarterExampleError(
                f"{example['id']} managed starter must run without a separate confirmation"
            )
        if "credits" not in route["required_confirmation"].lower():
            raise StarterExampleError(
                f"{example['id']} managed starter must report consumed credits with the result"
            )
        if not route["may_incur_cost"]:
            raise StarterExampleError(
                f"{example['id']} managed starter must still declare that it may incur cost"
            )
    elif route["may_incur_cost"]:
        if route["prompt_authorizes_execution"] is not False:
            raise StarterExampleError(
                f"{example['id']} remote or bulk starter must never authorize execution"
            )
        if route["confirmation_boundary"] != "required_before_paid_or_remote_execution":
            raise StarterExampleError(
                f"{example['id']} must require confirmation before remote or bulk execution"
            )
        confirmation = route["required_confirmation"].lower()
        for term in ("separate", "current-turn", "scope and cost preview"):
            if term not in confirmation:
                raise StarterExampleError(
                    f"{example['id']} must require separate current-turn confirmation after scope and cost preview"
                )
    else:
        if route["confirmation_boundary"] != "not_required_for_public_alpha_read":
            raise StarterExampleError(
                f"{example['id']} public read must state its exact confirmation boundary"
            )
        if route["prompt_authorizes_execution"] is not True:
            raise StarterExampleError(f"{example['id']} public-read authorization must be explicit")
    if route != EXPECTED_ROUTES[example["id"]]:
        raise StarterExampleError(
            f"{example['id']}.route must exactly match its pinned model, provider, and confirmation policy"
        )


def _validate_managed_workflow(example: dict[str, Any]) -> None:
    """Managed tutorial-scale work must run, and must check access before planning.

    This previously required the inverse: a scope/cost preview, then a separate
    confirmation, then execution. That turned every starter into a multi-turn
    approval workflow for calls costing fractions of a cent, and a user with no
    credential only discovered it after the plan.
    """

    if example["route"]["provider"] != "biohub-managed":
        return
    steps = [step.lower() for step in example["ordered_workflow"]]
    workflow = " ".join(steps)
    # Each entry is a set of acceptable spellings; "exactly once" reads better for a
    # command than "exactly one request" does.
    required_terms = (
        ("preflight",),
        ("exactly one", "exactly once"),
        ("without implicit retries",),
        ("credits",),
    )
    missing = [
        alternatives[0]
        for alternatives in required_terms
        if not any(term in workflow for term in alternatives)
    ]
    if missing:
        raise StarterExampleError(
            f"{example['id']} managed workflow must preflight, make exactly one "
            f"no-retry request, and report consumed credits (missing: {', '.join(missing)})"
        )
    for prohibited in (
        "separate explicit current-turn confirmation",
        "scope and cost preview",
        "otherwise stop",
    ):
        if prohibited in workflow:
            raise StarterExampleError(
                f"{example['id']} managed workflow must not gate execution on {prohibited}"
            )
    if "silently" in workflow:
        raise StarterExampleError(f"{example['id']} managed workflow must not conceal its actions")
    preflight_index = next(
        (index for index, step in enumerate(steps) if "preflight" in step),
        -1,
    )
    request_index = next(
        (index for index, step in enumerate(steps) if "without implicit retries" in step),
        -1,
    )
    presentation_index = next(
        (index for index, step in enumerate(steps) if "credits" in step),
        -1,
    )
    if not 0 == preflight_index < request_index < presentation_index:
        raise StarterExampleError(
            f"{example['id']} workflow must preflight first, then request, then present"
        )
    if tuple(example["ordered_workflow"]) != EXPECTED_MANAGED_WORKFLOWS[example["id"]]:
        raise StarterExampleError(
            f"{example['id']} managed workflow must exactly match its pinned step sequence"
        )


def _validate_example(example: dict[str, Any], sequence_record: dict[str, Any]) -> None:
    example_id = _nonempty_string(example.get("id"), "example.id")
    if example_id not in EXPECTED_EXAMPLES:
        raise StarterExampleError(f"unexpected starter example id: {example_id}")
    prompt = _nonempty_string(example.get("prompt"), f"{example_id}.prompt")
    if len(prompt) > 128:
        raise StarterExampleError(f"{example_id}.prompt exceeds 128 characters")
    if sequence_record["literal"] in prompt:
        raise StarterExampleError(
            f"{example_id}.prompt must not expose the hidden literal sequence fixture"
        )
    if PLACEHOLDER_PATTERN.search(prompt):
        raise StarterExampleError(f"{example_id}.prompt contains a placeholder")
    if prompt != EXPECTED_PROMPTS[example_id]:
        raise StarterExampleError(
            f"{example_id}.prompt must exactly match its natural-language launcher"
        )

    _validate_route(example)
    input_validation = _object(example.get("input_validation"), f"{example_id}.input_validation")
    if input_validation.get("sequence_id") != sequence_record["id"]:
        raise StarterExampleError(f"{example_id} references the wrong sequence id")
    if input_validation.get("alphabet") != CANONICAL_AMINO_ACIDS:
        raise StarterExampleError(f"{example_id} must pin the canonical amino-acid alphabet")
    maximum = input_validation.get("maximum_sequence_length")
    if maximum is None:
        _nonempty_string(
            input_validation.get("length_policy"),
            f"{example_id}.input_validation.length_policy",
        )
    else:
        if not isinstance(maximum, int) or isinstance(maximum, bool) or maximum <= 0:
            raise StarterExampleError(
                f"{example_id}.input_validation.maximum_sequence_length must be positive or null"
            )
        if sequence_record["length"] > maximum:
            raise StarterExampleError(f"{example_id} sequence exceeds its route-specific maximum")
    _string_list(
        input_validation.get("checks"),
        f"{example_id}.input_validation.checks",
        minimum=2,
    )
    _validate_mutations(example, sequence_record["literal"], input_validation)

    _string_list(example.get("ordered_workflow"), f"{example_id}.ordered_workflow", minimum=4)
    _validate_managed_workflow(example)
    artifacts = example.get("artifacts")
    if not isinstance(artifacts, list) or len(artifacts) < 2:
        raise StarterExampleError(f"{example_id}.artifacts must contain at least two artifacts")
    for index, artifact_value in enumerate(artifacts):
        artifact = _object(artifact_value, f"{example_id}.artifacts[{index}]")
        for field in ("name", "format"):
            _nonempty_string(artifact.get(field), f"{example_id}.artifacts[{index}].{field}")
        if not isinstance(artifact.get("required"), bool):
            raise StarterExampleError(f"{example_id}.artifacts[{index}].required must be boolean")
    _string_list(example.get("results"), f"{example_id}.results", minimum=2)
    _string_list(example.get("provenance"), f"{example_id}.provenance", minimum=2)
    _string_list(example.get("limits"), f"{example_id}.limits", minimum=2)

    presentation = _object(example.get("presentation"), f"{example_id}.presentation")
    for field in ("mode", "capability", "primary_artifact", "style", "fallback"):
        _nonempty_string(presentation.get(field), f"{example_id}.presentation.{field}")
    if presentation.get("automatic") is not True:
        raise StarterExampleError(f"{example_id}.presentation must be automatic")
    if presentation["mode"] == "structure-viewer" and presentation.get("contract") != (
        "structure_viewer_contract"
    ):
        raise StarterExampleError(f"{example_id}.presentation must reference the viewer contract")
    if presentation["mode"] != "structure-viewer" and "contract" in presentation:
        raise StarterExampleError(
            f"{example_id}.presentation must not claim the fold viewer contract"
        )
    if presentation != EXPECTED_PRESENTATIONS[example_id]:
        raise StarterExampleError(f"{example_id}.presentation contract has drifted")

    failure = _object(
        example.get("failure_and_nondeterminism"),
        f"{example_id}.failure_and_nondeterminism",
    )
    _string_list(
        failure.get("failure_modes"),
        f"{example_id}.failure_and_nondeterminism.failure_modes",
        minimum=2,
    )
    for field in ("nondeterminism", "retry_policy"):
        _nonempty_string(failure.get(field), f"{example_id}.failure_and_nondeterminism.{field}")
    qualification_gates = _string_list(
        example.get("qualification_gates"),
        f"{example_id}.qualification_gates",
        minimum=2,
    )
    qualification_text = " ".join(qualification_gates).lower()
    if "no-provider" not in qualification_text:
        raise StarterExampleError(f"{example_id} qualification must include a no-provider trace")
    if example["route"]["provider"] == "biohub-managed":
        for term in ("scope/cost preview", "separate explicit current-turn confirmation"):
            if term not in qualification_text:
                raise StarterExampleError(
                    f"{example_id} qualification must prove the managed confirmation stop"
                )


def _validate_execution_contracts(by_skill: dict[str, dict[str, Any]], sequence: str) -> None:
    esmc_example = by_skill["esmc"]
    if esmc_example.get("qualification_response") != EXPECTED_ESMC_QUALIFICATION_RESPONSE:
        raise StarterExampleError(
            "ESMC starter qualification response must match its exact eight-line contract"
        )
    esmc_execution = _object(
        esmc_example.get("execution_contract"),
        "gb1-esmc-w43f-masked-llr.execution_contract",
    )
    if esmc_execution != EXPECTED_ESMC_EXECUTION:
        raise StarterExampleError(
            "ESMC starter execution contract must pin one exact logits request, command, and output"
        )
    if validate_esmc_sequence(esmc_execution["sequence"]) != sequence:
        raise StarterExampleError("ESMC starter request changed the literal sequence")

    fold_example = by_skill["esmfold2"]
    if fold_example.get("qualification_response") != EXPECTED_FOLD_QUALIFICATION_RESPONSE:
        raise StarterExampleError(
            "fold starter qualification response must match its exact ten-line contract"
        )
    fold_execution = _object(
        fold_example.get("execution_contract"),
        "gb1-esmfold2-fast-fold.execution_contract",
    )
    if fold_execution != EXPECTED_FOLD_EXECUTION:
        raise StarterExampleError(
            "fold starter execution contract must pin one exact /fold request, command, and output"
        )
    fold_request = fold_execution["request"]
    fold_config = {
        key: value for key, value in fold_request.items() if key not in {"model", "sequence", "msa"}
    }
    try:
        if validate_esmc_sequence(fold_request["sequence"]) != sequence:
            raise StarterExampleError("fold starter request changed the literal sequence")
        validated_config = validate_fold_config(
            fold_config,
            model=fold_request["model"],
            endpoint="fold",
        )
    except ValidationError as exc:
        raise StarterExampleError(
            f"fold starter request failed production validation: {exc}"
        ) from exc
    if validated_config != fold_config or fold_request["include_pae"] is not True:
        raise StarterExampleError("fold starter must request pAE with exact validated parameters")

    atlas_example = by_skill["esm-atlas"]
    if atlas_example.get("qualification_response") != EXPECTED_ATLAS_QUALIFICATION_RESPONSE:
        raise StarterExampleError(
            "Atlas starter qualification response must match its exact eleven-line contract"
        )
    atlas_execution = _object(
        atlas_example.get("execution_contract"),
        "gb1-atlas-similarity-search.execution_contract",
    )
    if atlas_execution != EXPECTED_ATLAS_EXECUTION:
        raise StarterExampleError(
            "Atlas starter execution contract must pin one search, zero detail follow-ups, and at most one conditional presentation request"
        )
    if atlas_example.get("optional_artifacts") != EXPECTED_ATLAS_OPTIONAL_ARTIFACTS:
        raise StarterExampleError("Atlas starter optional search artifacts have drifted")
    workflow = " ".join(atlas_example["ordered_workflow"]).lower()
    if "fetch protein" in workflow or "representative-cluster" in workflow:
        raise StarterExampleError("Atlas starter must not reintroduce protein or cluster fetches")


def _validate_against_production_contracts(examples: list[dict[str, Any]], sequence: str) -> None:
    """Bind starter semantics to the production validators, router, and activator."""

    by_skill = {example["route"]["skill"]: example for example in examples}
    _validate_execution_contracts(by_skill, sequence)
    try:
        if validate_esmc_sequence(sequence) != sequence:
            raise StarterExampleError("production ESMC validation changed the literal sequence")
        if validate_atlas_search_sequence(sequence) != sequence:
            raise StarterExampleError("production Atlas validation changed the literal sequence")
        fold_input = validate_fold_input(
            {"sequences": [{"type": "protein", "id": "A", "sequence": sequence, "msa": None}]},
            model="esmfold2-fast-2026-05",
        )
    except ValidationError as exc:
        raise StarterExampleError(f"production input validation rejected a starter: {exc}") from exc
    if fold_input["sequences"][0]["sequence"] != sequence:
        raise StarterExampleError("production fold validation changed the literal sequence")

    for example_id, (
        request,
        expected_route,
        expected_model,
        expected_authentication,
    ) in EXPECTED_PRODUCTION_ROUTES.items():
        expected_declared = EXPECTED_ROUTES[example_id]
        skill = expected_declared["skill"]
        example = by_skill[skill]
        if select_skill(example["prompt"]) != skill:
            raise StarterExampleError(f"production activation drifted for {example['id']}")
        routed = route_request(request)
        if (routed.route, routed.model, routed.required_authentication) != (
            expected_route,
            expected_model,
            expected_authentication,
        ):
            raise StarterExampleError(f"production routing drifted for {example['id']}")
        declared_model = routed.model or "not-applicable-public-data-api"
        declared_provider = EXPECTED_PROVIDER_BY_PRODUCTION_ROUTE.get(routed.route)
        if (example["route"]["model"], example["route"]["provider"]) != (
            declared_model,
            declared_provider,
        ):
            raise StarterExampleError(
                f"declared route is inconsistent with production routing for {example['id']}"
            )
        artifact_contract = tuple(dict(artifact) for artifact in example["artifacts"])
        if artifact_contract != EXPECTED_ARTIFACTS[example_id]:
            raise StarterExampleError(
                f"artifact contract drifted for {example['id']}: names, formats, order, and requiredness are pinned"
            )
    atlas_example = by_skill["esm-atlas"]
    atlas_workflow = " ".join(atlas_example["ordered_workflow"])
    atlas_results = " ".join(atlas_example["results"])
    if "topk_results=10" not in atlas_workflow:
        raise StarterExampleError("Atlas starter must request topk_results=10")
    if "include_cluster_info=true" not in atlas_workflow:
        raise StarterExampleError("Atlas starter must request include_cluster_info=true")
    if "return up to ten actual ranked hits" not in atlas_workflow:
        raise StarterExampleError("Atlas starter must return up to ten actual ranked hits")
    if "zero protein or cluster-detail follow-up requests" not in atlas_workflow:
        raise StarterExampleError(
            "Atlas starter must make zero protein or cluster-detail follow-ups"
        )
    if "at most one public plddt thumbnail get" not in atlas_workflow.lower():
        raise StarterExampleError(
            "Atlas starter may make at most one conditional pLDDT thumbnail request"
        )
    if "up to ten actual returned hits" not in atlas_results.lower():
        raise StarterExampleError("Atlas starter results must allow up to ten actual returned hits")
    if "exactly ten" in f"{atlas_workflow} {atlas_results}".lower():
        raise StarterExampleError("Atlas starter must not require exactly ten returned hits")


def validate_starter_examples(contract: dict[str, Any]) -> dict[str, Any]:
    """Validate the self-contained contract without consulting repository surfaces."""

    if contract.get("schema_version") != "1.0":
        raise StarterExampleError("unsupported starter example schema_version")
    _validate_structure_viewer_contract(contract)
    if contract.get("prompt_max_characters") != 128:
        raise StarterExampleError("prompt_max_characters must be exactly 128")
    if contract.get("qualificationStatus") != "pending-clean-host-qualification":
        raise StarterExampleError(
            "qualificationStatus must remain pending-clean-host-qualification"
        )
    qualification = _object(contract.get("qualification"), "qualification")
    if qualification.get("cleanInstalledHostQualified") is not False:
        raise StarterExampleError("clean installed-host qualification must not be claimed")
    _nonempty_string(qualification.get("claim"), "qualification.claim")
    _string_list(qualification.get("gates"), "qualification.gates", minimum=3)
    sequence, _ = _validate_sequence(contract)
    surface_contract = _object(contract.get("surface_contract"), "surface_contract")
    for field in (
        "readme",
        "router_skill",
        "routing_reference",
        "activation_fixture",
    ):
        _nonempty_string(surface_contract.get(field), f"surface_contract.{field}")
    agent_defaults = _object(
        surface_contract.get("agent_defaults"), "surface_contract.agent_defaults"
    )
    if set(agent_defaults) != set(EXPECTED_EXAMPLES):
        raise StarterExampleError("surface_contract.agent_defaults does not cover every example")
    for example_id, path in agent_defaults.items():
        _nonempty_string(path, f"surface_contract.agent_defaults.{example_id}")

    examples = contract.get("examples")
    if not isinstance(examples, list) or len(examples) != len(EXPECTED_EXAMPLES):
        raise StarterExampleError("contract must contain exactly three starter examples")
    ids = [example.get("id") if isinstance(example, dict) else None for example in examples]
    if len(set(ids)) != len(ids) or set(ids) != set(EXPECTED_EXAMPLES):
        raise StarterExampleError("starter example ids must be unique and complete")
    for value in examples:
        _validate_example(_object(value, "example"), sequence)
    _validate_against_production_contracts(examples, sequence["literal"])
    return contract


def _read_default_prompt(path: Path) -> str:
    match = re.search(
        r"^\s*default_prompt:\s*(.+?)\s*$",
        path.read_text(encoding="utf-8"),
        re.MULTILINE,
    )
    if match is None:
        raise StarterExampleError(f"missing default_prompt in {path}")
    try:
        value = json.loads(match.group(1))
    except json.JSONDecodeError as exc:
        raise StarterExampleError(f"default_prompt in {path} must be JSON-quoted") from exc
    return _nonempty_string(value, f"default_prompt in {path}")


def validate_starter_example_surfaces(
    plugin_root: Path, contract: dict[str, Any]
) -> dict[str, Any]:
    """Reject prompt drift across the packaged focused-skill and activation surfaces."""

    validate_starter_examples(contract)
    plugin_root = plugin_root.resolve()
    surfaces = contract["surface_contract"]
    paths: dict[str, Path] = {}
    for field in (
        "readme",
        "router_skill",
        "routing_reference",
        "activation_fixture",
    ):
        path = (plugin_root / surfaces[field]).resolve()
        if not path.is_relative_to(plugin_root) or not path.is_file():
            raise StarterExampleError(f"surface {field} is missing or outside the plugin")
        paths[field] = path

    examples = contract["examples"]
    prompts = [example["prompt"] for example in examples]
    for field in ("readme", "router_skill", "routing_reference"):
        text = paths[field].read_text(encoding="utf-8")
        missing = [prompt for prompt in prompts if prompt not in text]
        if missing:
            raise StarterExampleError(f"{field} is missing {len(missing)} exact starter prompt(s)")

    fold_example = next(
        example for example in examples if example["id"] == "gb1-esmfold2-fast-fold"
    )
    request_path = (plugin_root / fold_example["execution_contract"]["request_file"]).resolve()
    if not request_path.is_relative_to(plugin_root) or not request_path.is_file():
        raise StarterExampleError("fold starter request file is missing or outside the plugin")
    try:
        request_payload = json.loads(request_path.read_text(encoding="utf-8"))
    except json.JSONDecodeError as exc:
        raise StarterExampleError("fold starter request file is not valid JSON") from exc
    if request_payload != EXPECTED_FOLD_EXECUTION["request"]:
        raise StarterExampleError("fold starter request file drifted from its execution contract")

    try:
        activation_cases = json.loads(paths["activation_fixture"].read_text(encoding="utf-8"))
    except json.JSONDecodeError as exc:
        raise StarterExampleError("activation fixture is not valid JSON") from exc
    if not isinstance(activation_cases, list):
        raise StarterExampleError("activation fixture must be a list")
    activation_by_prompt = {
        case.get("prompt"): case.get("expected")
        for case in activation_cases
        if isinstance(case, dict)
    }
    for example in examples:
        if activation_by_prompt.get(example["prompt"]) != example["route"]["skill"]:
            raise StarterExampleError(f"activation fixture has drifted for {example['id']}")

    for example_id, relative in surfaces["agent_defaults"].items():
        path = (plugin_root / relative).resolve()
        if not path.is_relative_to(plugin_root) or not path.is_file():
            raise StarterExampleError(f"agent default for {example_id} is missing")
        expected = next(example["prompt"] for example in examples if example["id"] == example_id)
        if _read_default_prompt(path) != expected:
            raise StarterExampleError(f"agent default has drifted for {example_id}")

    public_readme = plugin_root / "PUBLIC_README.md"
    if public_readme.exists():
        text = public_readme.read_text(encoding="utf-8")
        if any(prompt not in text for prompt in prompts):
            raise StarterExampleError("PUBLIC_README.md is missing exact starter prompts")
    return contract


def resolve_marketplace_default_prompts(
    contract: dict[str, Any],
) -> list[dict[str, str]]:
    """Resolve and validate the tutorial prompt records selected for the plugin page."""

    if contract.get("schema_version") != "1.1":
        raise StarterExampleError("unsupported tutorial use-case schema_version")
    _validate_structure_viewer_contract(contract)
    use_cases = contract.get("use_cases")
    if not isinstance(use_cases, list) or not use_cases:
        raise StarterExampleError("tutorial use_cases must be a non-empty list")

    prompts_by_id: dict[str, dict[str, str]] = {}
    prompt_ids_by_text: dict[str, str] = {}
    for use_case_index, use_case_value in enumerate(use_cases):
        use_case = _object(use_case_value, f"use_cases[{use_case_index}]")
        use_case_id = _nonempty_string(use_case.get("id"), f"use_cases[{use_case_index}].id")
        route = _object(use_case.get("route"), f"{use_case_id}.route")
        skill = _nonempty_string(route.get("skill"), f"{use_case_id}.route.skill")
        prompt_records = use_case.get("prompts")
        if not isinstance(prompt_records, list) or not prompt_records:
            raise StarterExampleError(f"{use_case_id}.prompts must be a non-empty list")
        for prompt_index, prompt_value in enumerate(prompt_records):
            prompt = _object(prompt_value, f"{use_case_id}.prompts[{prompt_index}]")
            prompt_id = _nonempty_string(
                prompt.get("id"), f"{use_case_id}.prompts[{prompt_index}].id"
            )
            text = _nonempty_string(
                prompt.get("text"), f"{use_case_id}.prompts[{prompt_index}].text"
            )
            if prompt_id in prompts_by_id:
                raise StarterExampleError(f"duplicate tutorial prompt id: {prompt_id}")
            if text in prompt_ids_by_text:
                raise StarterExampleError(
                    f"duplicate tutorial prompt text: {prompt_ids_by_text[text]} and {prompt_id}"
                )
            if text != " ".join(text.split()):
                raise StarterExampleError(f"{prompt_id} must use normalized whitespace")
            if len(text) > 128:
                raise StarterExampleError(f"{prompt_id} exceeds 128 characters")
            if MARKETPLACE_SEQUENCE_LITERAL_PATTERN.search(text):
                raise StarterExampleError(f"{prompt_id} exposes an amino-acid sequence")
            if PLACEHOLDER_PATTERN.search(text):
                raise StarterExampleError(f"{prompt_id} contains a placeholder")
            if "$" in text or MARKETPLACE_INVOCATION_PATTERN.search(text):
                raise StarterExampleError(f"{prompt_id} must not include invocation syntax")
            if select_skill(text) != skill:
                raise StarterExampleError(f"production activation drifted for {prompt_id}")
            prompts_by_id[prompt_id] = {
                "id": prompt_id,
                "text": text,
                "skill": skill,
                "use_case_id": use_case_id,
            }
            prompt_ids_by_text[text] = prompt_id

    default_ids = _string_list(
        contract.get("marketplace_default_prompt_ids"),
        "marketplace_default_prompt_ids",
        minimum=3,
    )
    if len(default_ids) != 3 or len(set(default_ids)) != 3:
        raise StarterExampleError(
            "marketplace_default_prompt_ids must contain exactly three unique prompt ids"
        )
    missing = [prompt_id for prompt_id in default_ids if prompt_id not in prompts_by_id]
    if missing:
        raise StarterExampleError(
            f"marketplace default prompt id is not defined: {', '.join(missing)}"
        )
    resolved = [prompts_by_id[prompt_id] for prompt_id in default_ids]
    if len({prompt["use_case_id"] for prompt in resolved}) != 3:
        raise StarterExampleError(
            "marketplace defaults must represent three distinct official tutorial workflows"
        )
    _validate_tutorial_execution_boundaries(contract, resolved)
    return resolved


def _validate_tutorial_execution_boundaries(
    contract: dict[str, Any], resolved: list[dict[str, str]]
) -> None:
    """Keep page-facing tutorial prompts fail-closed at target disclosure and spend consent."""

    policy = _object(contract.get("target_resolution_policy"), "target_resolution_policy")
    marketplace_policy = _nonempty_string(
        policy.get("marketplace_defaults"),
        "target_resolution_policy.marketplace_defaults",
    ).lower()
    # Target disclosure always matters. Execution authorization is deliberately
    # use-case-specific: PETase has a shipped replay-safe runner, while ATP/GLP do not.
    for term in (
        "disclose the exact pinned tutorial target or construct",
        "does not authorize target adoption",
    ):
        if term not in marketplace_policy:
            raise StarterExampleError(
                "marketplace default target policy must require exact target disclosure"
            )

    use_cases = {
        _nonempty_string(value.get("id"), "tutorial use-case id"): _object(
            value, "tutorial use case"
        )
        for value in contract["use_cases"]
        if isinstance(value, dict)
    }
    selected_use_case_ids = {prompt["use_case_id"] for prompt in resolved}
    for use_case_id in selected_use_case_ids:
        use_case = use_cases[use_case_id]
        execution_policy = " ".join(
            _string_list(use_case.get("execution_policy"), f"{use_case_id}.execution_policy")
        ).lower()
        if use_case_id == "esmc-mutation-landscape":
            for term in ("exact 259-request managed runtime", "indeterminate"):
                if term not in execution_policy:
                    raise StarterExampleError(
                        "the PETase execution policy must bind its exact replay-safe runtime"
                    )
            if "current-turn confirmation" in execution_policy:
                raise StarterExampleError(
                    "the PETase launcher must not contradict its exact managed runtime"
                )
        else:
            for term in (
                "does not authorize any provider call",
                "explicit current-turn confirmation",
            ):
                if term not in execution_policy:
                    raise StarterExampleError(
                        f"{use_case_id} must require explicit confirmation before provider calls"
                    )

    fold_use_case = use_cases["esmfold2-all-atom-and-msa"]
    if fold_use_case.get("presentation_contract") != "structure_viewer_contract":
        raise StarterExampleError(
            "esmfold2-all-atom-and-msa must reference structure_viewer_contract"
        )
    sae_use_case = use_cases["esmc-sae-feature-interpretation"]
    if "presentation_contract" in sae_use_case:
        raise StarterExampleError(
            "esmc-sae-feature-interpretation must not claim the fold presentation contract"
        )

    mutation = use_cases.get("esmc-mutation-landscape")
    if mutation is None:
        raise StarterExampleError("esmc-mutation-landscape tutorial contract is missing")
    mutation_route = _object(mutation.get("route"), "esmc-mutation-landscape.route")
    # The pinned notebook runs all 259 contexts against the managed API. Routing them
    # to Modal pointed at a function the plugin never ships.
    if mutation_route.get("execution_route") != "biohub":
        raise StarterExampleError("the PETase landscape must route to the managed Biohub API")
    if mutation_route.get("model_id") != "esmc-600m-2024-12":
        raise StarterExampleError("the PETase landscape must use the notebook's pinned model")
    if mutation_route.get("masked_context_count") != 259:
        raise StarterExampleError("the PETase landscape must disclose 259 masked contexts")
    concurrency = _nonempty_string(
        mutation_route.get("concurrency"), "esmc-mutation-landscape.route.concurrency"
    ).lower()
    for term in ("threadpoolexecutor", "bounded", "host-pinned"):
        if term not in concurrency:
            raise StarterExampleError(
                "the PETase landscape must run concurrent managed calls through a bounded, "
                "host-pinned pool as the quickstart documents"
            )
    mutation_runtime = _object(mutation.get("runtime"), "esmc-mutation-landscape.runtime")
    resume_command = mutation_runtime.get("resume_command")
    if (
        not isinstance(resume_command, list)
        or "--resume" not in resume_command
        or "<existing-output-dir>" not in resume_command
    ):
        raise StarterExampleError("the PETase runtime must publish its explicit resume command")
    checkpoint_contract = _nonempty_string(
        mutation_runtime.get("checkpoint_contract"),
        "esmc-mutation-landscape.runtime.checkpoint_contract",
    ).lower()
    for term in ("exact-request-bound", "retry-after", "indeterminate"):
        if term not in checkpoint_contract:
            raise StarterExampleError(
                "the PETase runtime must checkpoint exact requests and refuse unsafe replay"
            )

    sae = use_cases.get("esmc-sae-feature-interpretation")
    sae_target = _object(sae.get("target") if sae else None, "SAE tutorial target")
    if (
        sae_target.get("pdb_id") != "2XND"
        or sae_target.get("chain_id") != "A"
        or sae_target.get("structure_evidence_type") != "experimental-reference"
    ):
        raise StarterExampleError(
            "the SAE marketplace prompt must disclose experimental PDB 2XND chain A"
        )
    sae_contract = _object(
        sae.get("reproducibility_contract") if sae else None,
        "SAE tutorial reproducibility contract",
    )
    expected_sae_contract = {
        "active_value_threshold": 0.01,
        "active_value_comparator": ">",
        "rank_top_k_by_max_activation": 10,
        "rank_top_k_by_prevalence": 10,
        "describe_top_k_by_max_activation": 5,
        "map_top_k_by_max_activation": 3,
        "managed_requests": {"encode": 1, "logits": 1},
        "public_data_requests": {
            "rcsb_fasta": 1,
            "rcsb_structure": 1,
            "atlas_feature_detail": 5,
        },
        "tokenization": "managed-encode",
        "normalize_features": True,
        "artifacts": [
            "2xnd-chain-a.fasta",
            "2xnd.pdb",
            "encode-raw-response.json",
            "logits-raw-response.json",
            "sae-features.npz",
            "feature-rankings.json",
            "atlas-feature-responses.json",
            "per-residue-activations.csv",
            "structure-viewer-handoff.json",
            "provenance.json",
        ],
    }
    if sae_contract != expected_sae_contract:
        raise StarterExampleError(
            "the SAE marketplace prompt must pin the official threshold, ranking, request, tokenization, and artifact contract"
        )

    fold = use_cases.get("esmfold2-all-atom-and-msa")
    fold_targets = _object(fold.get("targets") if fold else None, "ESMFold2 tutorial targets")
    modified = _object(
        fold_targets.get("modified_glp1r_peptide_linker"),
        "modified GLP-1R tutorial target",
    )
    receptor = _object(modified.get("receptor"), "modified GLP-1R receptor")
    linker = _object(modified.get("linker"), "modified GLP-1R linker")
    if (
        "not canonical untagged glp1r"
        not in _nonempty_string(
            receptor.get("construct_note"), "modified GLP-1R construct note"
        ).lower()
    ):
        raise StarterExampleError(
            "the GLP-1R marketplace prompt must disclose the tagged construct"
        )
    if (
        "not the exact therapeutic linker"
        not in _nonempty_string(linker.get("identity"), "modified GLP-1R linker identity").lower()
    ):
        raise StarterExampleError(
            "the GLP-1R marketplace prompt must disclose the representative linker"
        )


def validate_tutorial_marketplace_surfaces(
    plugin_root: Path, contract: dict[str, Any]
) -> list[dict[str, str]]:
    """Reject drift between the tutorial prompt owner and every plugin-page surface."""

    resolved = resolve_marketplace_default_prompts(contract)
    plugin_root = plugin_root.resolve()
    surfaces = _object(contract.get("surface_contract"), "tutorial surface_contract")
    paths: dict[str, Path] = {}
    for field in (
        "manifest",
        "readme",
        "router_skill",
        "routing_reference",
        "activation_fixture",
    ):
        relative = _nonempty_string(surfaces.get(field), f"surface_contract.{field}")
        path = (plugin_root / relative).resolve()
        if not path.is_relative_to(plugin_root) or not path.is_file():
            raise StarterExampleError(f"tutorial surface {field} is missing or outside the plugin")
        paths[field] = path

    manifest = _load_json_object(paths["manifest"], "plugin manifest")
    expected_texts = [prompt["text"] for prompt in resolved]
    if manifest.get("interface", {}).get("defaultPrompt") != expected_texts:
        raise StarterExampleError(
            "manifest defaultPrompt has drifted from tutorial marketplace defaults"
        )

    for field in ("readme", "router_skill", "routing_reference"):
        text = paths[field].read_text(encoding="utf-8")
        missing = [prompt for prompt in expected_texts if prompt not in text]
        if missing:
            raise StarterExampleError(
                f"{field} is missing {len(missing)} tutorial marketplace prompt(s)"
            )

    specialist_skills = _object(
        surfaces.get("specialist_skills"), "surface_contract.specialist_skills"
    )
    expected_specialists = {prompt["skill"] for prompt in resolved}
    if set(specialist_skills) != expected_specialists:
        raise StarterExampleError(
            "surface_contract.specialist_skills must cover every marketplace specialist"
        )
    for skill, relative_value in specialist_skills.items():
        relative = _nonempty_string(relative_value, f"surface_contract.specialist_skills.{skill}")
        path = (plugin_root / relative).resolve()
        if not path.is_relative_to(plugin_root) or not path.is_file():
            raise StarterExampleError(f"tutorial specialist surface for {skill} is missing")
        text = path.read_text(encoding="utf-8")
        missing = [
            prompt["text"]
            for prompt in resolved
            if prompt["skill"] == skill and prompt["text"] not in text
        ]
        if missing:
            raise StarterExampleError(
                f"specialist surface {skill} is missing {len(missing)} marketplace prompt(s)"
            )

    try:
        activation_cases = json.loads(paths["activation_fixture"].read_text(encoding="utf-8"))
    except json.JSONDecodeError as exc:
        raise StarterExampleError("activation fixture is not valid JSON") from exc
    if not isinstance(activation_cases, list):
        raise StarterExampleError("activation fixture must be a list")
    activation_by_prompt = {
        case.get("prompt"): case.get("expected")
        for case in activation_cases
        if isinstance(case, dict)
    }
    for prompt in resolved:
        if activation_by_prompt.get(prompt["text"]) != prompt["skill"]:
            raise StarterExampleError(
                f"activation fixture has drifted for tutorial prompt {prompt['id']}"
            )
    return resolved


def validate_starter_examples_file(plugin_root: Path) -> dict[str, Any]:
    path = plugin_root / "examples" / "starter-examples.json"
    validated = validate_starter_example_surfaces(plugin_root, load_starter_examples(path))
    tutorial_path = plugin_root / "examples" / "tutorial-use-cases.json"
    validate_tutorial_marketplace_surfaces(
        plugin_root,
        load_tutorial_use_cases(tutorial_path),
    )
    return validated

SHA-256: f2ba7af140fba077e425ed8808fd05b385258f82e58d5e74b9ba0fad32b8d14c