← Files AkinatorARCHIVED FILE

skills/everything/scripts/akinator_coverage.py

44.8 KB · Oct 5, 2026 · 18:32 UTC

↓ Download file

#!/usr/bin/env python3
"""Akinator coverage checker - the mechanically verifiable knowledge invariants.

Runs the Part 14.1 invariants against a repository and reports every violation
ranked by severity. Designed to run in CI and on demand.

NEVER wire this into a git hook. Git hooks gate code and must stay fast;
knowledge enforcement lives in CI, in the test suite and in session behavior.
See rules/05-no-git-hook-complication.md.

Stdlib only, deterministic output, exit code driven:
    0  no CRITICAL and no HIGH findings (or --strict satisfied)
    1  findings at or above the failure threshold
    2  the checker itself could not run

Usage:
    python skills/everything/scripts/akinator_coverage.py [REPO_ROOT] [options]

Options:
    --json                 emit machine-readable findings instead of a report
    --strict               fail on MEDIUM findings too
    --fail-on LEVEL        critical | high | medium | low  (default: high)
    --only CHECK[,CHECK]   run only the named checks (see --list-checks)
    --skip CHECK[,CHECK]   skip the named checks
    --list-checks          print the check ids and exit
    --sample N             max references sampled per doc for truth checking
                           (default 40; 0 means all)
"""

from __future__ import annotations

import argparse
import fnmatch
import json
import os
import re
import sys
from dataclasses import dataclass, asdict, field
from pathlib import Path
from typing import Iterable, Sequence

# --------------------------------------------------------------------------
# Severity
# --------------------------------------------------------------------------

SEVERITIES = ("critical", "high", "medium", "low")
_SEV_RANK = {name: index for index, name in enumerate(SEVERITIES)}


@dataclass
class Finding:
    check: str
    severity: str
    path: str
    message: str
    fix: str = ""

    def sort_key(self) -> tuple:
        return (_SEV_RANK[self.severity], self.check, self.path, self.message)


# --------------------------------------------------------------------------
# Repository model
# --------------------------------------------------------------------------

ROUTER_NAMES = (
    "CLAUDE.md",
    "AGENTS.md",
    "CODEX.md",
    "GEMINI.md",
    "GLM.md",
    "KIMI.md",
    "QWEN.md",
    "MISTRAL.md",
    "DEEPSEEK.md",
    ".cursorrules",
)

# Directories that never hold knowledge artifacts and must not be walked.
IGNORED_DIRS = {
    ".git", "node_modules", "__pycache__", ".venv", "venv", "dist", "build",
    ".next", ".nuxt", "target", "vendor", ".pytest_cache", ".mypy_cache",
    ".tox", "coverage", ".idea", ".vscode", ".gradle", "bin", "obj",
}

KNOWLEDGE_DIRS = ("rules", "skills", "context", "memory", "docs")

# Files that mark a directory as a module or service worth its own router.
MODULE_MANIFESTS = (
    "package.json", "pyproject.toml", "setup.py", "go.mod", "Cargo.toml",
    "composer.json", "build.gradle", "build.gradle.kts", "pom.xml",
    "Gemfile", "mix.exs",
)

GIT_HOOK_MARKERS = (
    ".git/hooks", ".husky", "pre-commit", "pre-push", "lint-staged",
    ".pre-commit-config",
)

GENERATED_MARKERS = (
    "do not edit", "do not hand-edit", "auto-generated", "autogenerated",
    "generated by", "generated file",
)

# A markdown link: [text](target)
MD_LINK = re.compile(r"\[[^\]]*\]\(([^)\s]+)(?:\s+\"[^\"]*\")?\)")
# A backticked token that looks like a repo path: has a slash and an extension,
# or is a known knowledge directory entry.
BACKTICK_PATH = re.compile(r"`([A-Za-z0-9_./\\-]+\.[A-Za-z0-9]{1,8})`")
# Headings, used to locate a rule's Enforcement section.
HEADING = re.compile(r"^#{1,6}\s+(.*?)\s*$", re.MULTILINE)
# A generated-file banner: an HTML comment at the very top, optionally after
# YAML frontmatter (Cursor's .mdc puts frontmatter first).
LEADING_COMMENT = re.compile(
    r"\A(?:---\n.*?\n---\n\s*)?<!--.*?-->", re.DOTALL
)
# An intentional per-tool divergence marker in a router.
TOOL_SPECIFIC = re.compile(r"<!--\s*akinator:tool-specific\s*-->", re.IGNORECASE)

# A vendored artifact declares its upstream and how to refresh it, INSTEAD of
# naming a local generator. Both halves are required: "installed from X" with no
# way to update is as useless to the reader as a generator that is not there.
VENDORED_ORIGIN = re.compile(r"\binstalled from\b", re.IGNORECASE)
VENDORED_UPDATE = re.compile(r"\breinstall\b", re.IGNORECASE)


IGNORE_FILE = ".akinatorignore"


def load_ignore_patterns(root: Path) -> list[str]:
    """Read `.akinatorignore`: one path prefix or fnmatch pattern per line.

    Exists for content that is deliberately not held to the invariants -
    fixture repositories with planted defects, vendored trees, generated
    scratch. Deliberately simple: no negation, no nesting. Anything more
    expressive would let a repository quietly exclude its way to green, which
    is the failure mode this whole checker is built against.
    """
    path = root / IGNORE_FILE
    if not path.is_file():
        return []
    patterns: list[str] = []
    for line in path.read_text(encoding="utf-8", errors="replace").splitlines():
        line = line.split("#", 1)[0].strip()
        if line:
            patterns.append(line.rstrip("/").replace("\\", "/"))
    return patterns


class Repo:
    """Everything the checks need to know about the tree, read once."""

    def __init__(self, root: Path, extra_ignores: Sequence[str] = ()) -> None:
        self.root = root
        self.ignore_patterns = load_ignore_patterns(root) + list(extra_ignores)
        self.files: list[Path] = []
        self.markdown: list[Path] = []
        self._text_cache: dict[Path, str] = {}
        self._scan()

    def is_ignored(self, relpath: str) -> bool:
        relpath = relpath.replace("\\", "/")
        for pattern in self.ignore_patterns:
            if relpath == pattern or relpath.startswith(pattern + "/"):
                return True
            if fnmatch.fnmatch(relpath, pattern):
                return True
        return False

    def _scan(self) -> None:
        for dirpath, dirnames, filenames in os.walk(self.root):
            dirnames[:] = sorted(
                d for d in dirnames
                if d not in IGNORED_DIRS
                and not self.is_ignored(
                    self.rel(Path(dirpath) / d)
                )
            )
            for name in sorted(filenames):
                path = Path(dirpath) / name
                if self.is_ignored(self.rel(path)):
                    continue
                self.files.append(path)
                # .mdc is Cursor's rule format. Leaving it out made every
                # Cursor rule invisible: never a router, never link-checked, and
                # a banner parser written for .mdc frontmatter that no file could
                # ever reach.
                if name.lower().endswith((".md", ".mdx", ".mdc")):
                    self.markdown.append(path)
        self.files.sort()
        self.markdown.sort()
        self._relative = {self.rel(p) for p in self.files}
        # Directories are legitimate link targets too.
        self._relative |= {
            self.rel(p) for p in {f.parent for f in self.files} if p != self.root
        }

    def rel(self, path: Path) -> str:
        try:
            return path.relative_to(self.root).as_posix()
        except ValueError:
            return path.as_posix()

    def text(self, path: Path) -> str:
        if path not in self._text_cache:
            try:
                self._text_cache[path] = path.read_text(
                    encoding="utf-8", errors="replace"
                )
            except OSError:
                self._text_cache[path] = ""
        return self._text_cache[path]

    def exists_rel(self, relpath: str) -> bool:
        # NOT lstrip("./") - that strips every leading '.' and '/', which turns
        # `.claude-plugin/plugin.json` into `claude-plugin/plugin.json` and
        # reports an existing dotfile as missing.
        relpath = relpath.strip().replace("\\", "/")
        while relpath.startswith("./"):
            relpath = relpath[2:]
        relpath = relpath.lstrip("/")
        if not relpath:
            return False
        return relpath in self._relative or (self.root / relpath).exists()

    # -- knowledge layer discovery -----------------------------------------

    def routers(self) -> list[Path]:
        found = [p for p in self.files if p.name in ROUTER_NAMES]
        # .cursor/rules/*.mdc (and .md) are routers too.
        cursor = self.root / ".cursor" / "rules"
        if cursor.is_dir():
            found += [p for p in self.markdown if cursor in p.parents]
        return sorted(set(found))

    def root_routers(self) -> list[Path]:
        return [p for p in self.routers() if p.parent == self.root]

    def knowledge_artifacts(self) -> list[Path]:
        out: list[Path] = []
        for directory in KNOWLEDGE_DIRS:
            base = self.root / directory
            if not base.is_dir():
                continue
            for path in self.markdown:
                if base in path.parents or path.parent == base:
                    out.append(path)
        return sorted(set(out))

    def rules(self) -> list[Path]:
        base = self.root / "rules"
        if not base.is_dir():
            return []
        return sorted(p for p in self.markdown if p.parent == base
                      and p.name.lower() not in ("readme.md", "index.md"))

    def skills(self) -> list[Path]:
        base = self.root / "skills"
        if not base.is_dir():
            return []
        return sorted(p for p in self.markdown if p.name == "SKILL.md"
                      and base in p.parents)

    def has_knowledge_layer(self) -> bool:
        return bool(self.routers()) or any(
            (self.root / d).is_dir() for d in KNOWLEDGE_DIRS
        )

    def modules(self) -> list[Path]:
        """Directories below the root that look like a module or service."""
        out: set[Path] = set()
        for path in self.files:
            if path.name in MODULE_MANIFESTS and path.parent != self.root:
                out.add(path.parent)
        return sorted(out)


# --------------------------------------------------------------------------
# Link extraction
# --------------------------------------------------------------------------

def _is_external(target: str) -> bool:
    lowered = target.lower()
    return (
        lowered.startswith(("http://", "https://", "mailto:", "tel:", "#"))
        or lowered.startswith("${")
    )


FENCE = re.compile(r"^[ \t]*(?:```|~~~).*?^[ \t]*(?:```|~~~)[ \t]*$",
                   re.MULTILINE | re.DOTALL)


def prose_of(repo: Repo, path: Path) -> str:
    """Document text with fenced code blocks removed.

    Everything inside a fence is an illustration - an example link, a sample
    path, a template skeleton. Treating those as claims about this tree
    produces false findings, so they are stripped before extraction.
    """
    return FENCE.sub("\n", repo.text(path))


def links_in(repo: Repo, path: Path) -> list[str]:
    """Repo-relative link targets referenced by a markdown file."""
    out: list[str] = []
    for target in MD_LINK.findall(prose_of(repo, path)):
        if _is_external(target):
            continue
        target = target.split("#", 1)[0].strip()
        if not target:
            continue
        out.append(_resolve(repo, path, target))
    return out


def _resolve(repo: Repo, source: Path, target: str) -> str:
    """Resolve a link target to a repo-relative posix path."""
    target = target.replace("\\", "/")
    if target.startswith("/"):
        return target.lstrip("/")
    resolved = (source.parent / target).resolve()
    try:
        return resolved.relative_to(repo.root).as_posix()
    except ValueError:
        return os.path.normpath(target).replace("\\", "/")


def path_mentions(repo: Repo, path: Path, limit: int) -> list[str]:
    """Backticked path-looking tokens in a doc, for truth checking."""
    seen: list[str] = []
    for token in BACKTICK_PATH.findall(prose_of(repo, path)):
        token = token.replace("\\", "/")
        if token in seen:
            continue
        seen.append(token)
        if limit and len(seen) >= limit:
            break
    return seen


# --------------------------------------------------------------------------
# Checks
# --------------------------------------------------------------------------

CHECKS: dict[str, str] = {
    "reachability": "Every rule, skill, context map, doc and memory entry is "
                    "reachable from an index or router.",
    "index-completeness": "Every artifact appears in its own category index, "
                          "not merely somewhere in the tree.",
    "dead-links": "No markdown link points at a file that does not exist.",
    "rule-enforcement": "Every rule names an enforcement mechanism that exists "
                        "in the tree, and it is not a git hook.",
    "router-sync": "Root routers do not fork - none references knowledge the "
                   "others omit without an intentional-divergence marker.",
    "module-routers": "Every module or service has a local router file.",
    "generated": "Generated artifacts name a generator that exists; vendored "
                 "artifacts name their origin and how to refresh them.",
    "doc-truth": "Paths named in docs exist in the tree.",
    "skill-format": "Every skill has frontmatter with name and description, and "
                    "the required sections.",
    "staleness": "Every context map and doc states what would make it stale.",
    "git-hooks": "No knowledge or documentation check is wired into a git hook.",
}


def check_reachability(repo: Repo) -> list[Finding]:
    artifacts = sorted(set(repo.knowledge_artifacts()) | set(repo.skills()))
    if not artifacts:
        return []

    referenced: set[str] = set()
    for source in repo.markdown:
        for target in links_in(repo, source):
            referenced.add(target)
            # A link to a skill directory reaches its SKILL.md.
            referenced.add(f"{target}/SKILL.md")
        for token in path_mentions(repo, source, 0):
            referenced.add(token)

    findings: list[Finding] = []
    for artifact in sorted(set(artifacts)):
        rel = repo.rel(artifact)
        if artifact.name.lower() in ("readme.md", "index.md"):
            continue  # indexes are entry points, not entries
        if rel in referenced:
            continue
        # A SKILL.md is also reachable via its directory name.
        if artifact.name == "SKILL.md" and repo.rel(artifact.parent) in referenced:
            continue
        findings.append(Finding(
            check="reachability",
            severity="medium",
            path=rel,
            message="not reachable from any index or router - unindexed means "
                    "nonexistent",
            fix="Add it to its category index and link that index from the "
                "routers (skill: akinator-index-sync).",
        ))
    return findings


# Where each kind of artifact is indexed, and the glob that finds its members.
# The index is the file a reader actually browses; being linked from somewhere
# else in the tree is not the same thing.
CATEGORY_INDEXES = (
    ("rules", "*.md", ("rules/README.md", "rules/index.md")),
    ("skills", "*/SKILL.md", ("docs/skills.md", "skills/README.md")),
    ("context", "*.md", ("context/README.md", "context/index.md")),
    ("memory", "*.md", ("memory/index.md", "memory/README.md")),
    ("docs/adr", "*.md", ("docs/adr/README.md", "docs/adr/index.md")),
    ("docs", "*.md", ("docs/README.md", "docs/index.md")),
    # The remaining taxonomy homes. These are where a target repo's business
    # rules, product intent and runbooks live - the categories most likely to
    # rot, and the ones an earlier revision silently exempted while the skill's
    # check table promised "every artifact".
    ("docs/business", "*.md", ("docs/business/README.md", "docs/business/index.md")),
    ("docs/product", "*.md", ("docs/product/README.md", "docs/product/index.md")),
    ("docs/ops", "*.md", ("docs/ops/README.md", "docs/ops/index.md")),
    ("templates", "*.md", ("templates/README.md", "templates/index.md")),
    ("agents", "*.md", ("docs/agents.md", "agents/README.md")),
    ("evals/suites", "*.md", ("evals/README.md",)),
)


def _normalize_rel(target: str) -> str:
    """Strip a leading `./` without eating a leading dot.

    NOT lstrip("./"), which strips every leading '.' and '/' and would turn
    `.claude-plugin/plugin.json` into `claude-plugin/plugin.json`. `exists_rel`
    carries the same warning; it was written after being bitten by exactly this.
    Here the consequence would be a silently widened accepted set - a false
    negative in the check whose whole lesson is that false negatives are silent.
    """
    target = target.strip().replace("\\", "/")
    while target.startswith("./"):
        target = target[2:]
    return target.lstrip("/")


def _indexed_paths(repo: Repo, index: Path) -> set[str]:
    """Every repo-relative path an index actually points at.

    Resolved, not pattern-matched. Three silent false negatives were found in
    review before this was rewritten to resolve paths:

        `akinator`     matched inside every `akinator-*` entry
        `demo`         matched inside a listed `demo-extended`
        `overview.md`  matched a link to `adr/overview.md`

    and the regex that closed the third produced six false *positives* on
    `evals/suites/*` in the same run, because those are listed with a directory
    prefix. Boundary-matching a bare token cannot separate "this artifact" from
    "a different artifact whose name contains it" - resolving the reference can.
    """
    text = prose_of(repo, index)
    out: set[str] = set()

    for target in MD_LINK.findall(text) + BACKTICK_PATH.findall(text):
        if _is_external(target):
            continue
        target = target.split("#", 1)[0].strip()
        if not target:
            continue
        # An index may reference a sibling relatively, or spell the full
        # repo-relative path. Both are legitimate; accept either reading.
        out.add(_resolve(repo, index, target))
        out.add(_normalize_rel(target))

    return out


def check_index_completeness(repo: Repo) -> list[Finding]:
    """Every artifact appears in its **own** category index, not merely somewhere.

    `reachability` proves an artifact is referenced from some markdown file. That
    is weaker than the taxonomy's actual law. An artifact linked only from a
    router, or only from a sibling doc, is invisible to a reader who opens the
    index for that category and reads down the list - which is exactly how a
    fresh agent looks for things.
    """
    findings: list[Finding] = []

    for directory, pattern, index_candidates in CATEGORY_INDEXES:
        base = repo.root / directory
        if not base.is_dir():
            continue

        index = next(
            (c for c in index_candidates if repo.exists_rel(c)), None
        )
        if index is None:
            continue  # no index for this category; `reachability` covers that

        listed = _indexed_paths(repo, repo.root / index)
        members = [
            p for p in sorted(base.glob(pattern))
            if p.name.lower() not in ("readme.md", "index.md")
            and not repo.is_ignored(repo.rel(p))
        ]

        for member in members:
            rel = repo.rel(member)
            # A skill may be listed by its directory or by its SKILL.md; both
            # resolve to a real path, so accept either.
            accepted = {rel}
            if member.name == "SKILL.md":
                accepted.add(repo.rel(member.parent))

            if accepted & listed:
                continue
            findings.append(Finding(
                check="index-completeness",
                severity="medium",
                path=rel,
                message=f"not listed in its category index ({index}) - a reader "
                        "browsing that index will not find it",
                fix=f"Add an entry to {index} naming the situation this artifact "
                    "serves, not just its title (skill: akinator-index-sync).",
            ))

    return findings


def check_dead_links(repo: Repo) -> list[Finding]:
    findings: list[Finding] = []
    for source in repo.markdown:
        for target in sorted(set(links_in(repo, source))):
            if repo.exists_rel(target):
                continue
            findings.append(Finding(
                check="dead-links",
                severity="high",
                path=repo.rel(source),
                message=f"links to '{target}', which does not exist",
                fix="Fix or remove the link. Dead links teach readers that "
                    "indexes cannot be trusted (skill: akinator-index-sync).",
            ))
    return findings


def _section(text: str, *names: str) -> str:
    """Return the body of the first heading whose title *is* one of `names`.

    Exact match, not substring. A rule titled "Every rule names an enforcement
    mechanism that exists" contains the word "enforcement" in its H1; a
    substring match would return that heading's (empty) body and report the rule
    as having no Enforcement section at all.
    """
    # Strip fences first: a rule's Prohibited-patterns block routinely contains
    # a fenced example with its own `## Enforcement` heading, and matching that
    # returns the example's body instead of the real section.
    text = FENCE.sub("\n", text)
    headings = list(HEADING.finditer(text))
    for index, match in enumerate(headings):
        title = match.group(1).strip().lower().strip("#*: ")
        if title in names:
            start = match.end()
            end = headings[index + 1].start() if index + 1 < len(headings) else len(text)
            return text[start:end]
    return ""


MECHANISM_LINE = re.compile(r"^\s*[-*]\s*mechanism\s*:", re.IGNORECASE)
BACKTICKED = re.compile(r"`([^`]+)`")


def _declared_mechanisms(enforcement_body: str) -> list[str]:
    """The mechanism paths a rule declares in its Enforcement section.

    The template's convention is `- Mechanism: <path> - explanation`, so the
    mechanism is the **first** backticked token on each `Mechanism:` line;
    everything after it is prose. Taking only the first token matters: rule 05's
    mechanism line goes on to name the hook files its checker *scans*, and those
    are not what enforces the rule.

    Falls back to every path-shaped token in the section for rules that predate
    the convention - a repo Akinator has just onboarded will have some.
    """
    declared: list[str] = []
    for line in enforcement_body.splitlines():
        if not MECHANISM_LINE.match(line):
            continue
        tokens = BACKTICKED.findall(line)
        if tokens:
            declared.append(tokens[0].replace("\\", "/").strip())

    if declared:
        return declared
    return [t.replace("\\", "/") for t in BACKTICK_PATH.findall(enforcement_body)]


def check_rule_enforcement(repo: Repo) -> list[Finding]:
    findings: list[Finding] = []
    for rule in repo.rules():
        rel = repo.rel(rule)
        text = repo.text(rule)
        body = _section(text, "enforcement")
        if not body.strip():
            findings.append(Finding(
                check="rule-enforcement",
                severity="high",
                path=rel,
                message="has no Enforcement section - a rule without a live "
                        "enforcement mechanism is not a rule",
                fix="Add an Enforcement section naming a test, lint rule, type "
                    "constraint or CI step that exists (skill: "
                    "akinator-rule-forge).",
            ))
            continue

        candidates = _declared_mechanisms(body)

        # Judge the declared mechanism paths, not the prose. A rule about git
        # hooks necessarily describes them; that is not the same as being
        # enforced by one.
        hooked = [
            t for t in candidates
            if any(marker in t.lower() for marker in GIT_HOOK_MARKERS)
        ]
        if hooked:
            findings.append(Finding(
                check="rule-enforcement",
                severity="critical",
                path=rel,
                message=f"enforces via a git hook ({', '.join(sorted(set(hooked)))}) "
                        "- prohibited; git hooks gate code, not knowledge",
                fix="Move enforcement to a test, a lint rule or a CI step "
                    "(rules/05-no-git-hook-complication.md).",
            ))
            candidates = [t for t in candidates if t not in hooked]
        # A mechanism may name a test by `path::test_name`; the path is what
        # must exist.
        existing = [t for t in candidates if repo.exists_rel(t.split("::", 1)[0])]
        if candidates and not existing:
            findings.append(Finding(
                check="rule-enforcement",
                severity="critical",
                path=rel,
                message="names an enforcement mechanism that does not exist in "
                        f"the tree: {', '.join(sorted(set(candidates))[:5])}",
                fix="Build the mechanism, or point the rule at one that exists. "
                    "A named-but-absent mechanism is fake compliance (skill: "
                    "akinator-anti-gaming).",
            ))
        elif not candidates:
            findings.append(Finding(
                check="rule-enforcement",
                severity="medium",
                path=rel,
                message="Enforcement section names no concrete mechanism path",
                fix="Name the test file, lint rule or CI step by path, in "
                    "backticks, so it can be verified.",
            ))
    return findings


def strip_tool_specific(text: str) -> tuple[str, bool]:
    """Remove tool-specific sections from a router's text.

    The marker `<!-- akinator:tool-specific -->` precedes a heading; the marked
    section runs from the marker to the next `##`-level heading, or to the end
    of the file.

    Content inside such a section is excluded from the sync comparison in both
    directions - it neither triggers a finding against its own router nor
    creates an expectation for the others. Counting it in the union would make a
    legitimately Codex-only note look like something every router is missing.
    """
    NORMAL, AWAIT_HEADING, IN_SECTION = 0, 1, 2
    state = NORMAL
    marked = False
    out: list[str] = []

    for line in text.splitlines(keepends=True):
        if state == NORMAL:
            if TOOL_SPECIFIC.search(line):
                marked = True
                state = AWAIT_HEADING
                continue
            out.append(line)
        elif state == AWAIT_HEADING:
            # Everything up to and including the marked heading is dropped.
            if line.startswith("## "):
                state = IN_SECTION
        else:  # IN_SECTION
            if line.startswith("## "):
                state = NORMAL
                out.append(line)

    return "".join(out), marked


def check_router_sync(repo: Repo) -> list[Finding]:
    routers = repo.root_routers()
    if len(routers) < 2:
        return []

    knowledge_links: dict[str, set[str]] = {}
    for router in routers:
        rel = repo.rel(router)
        shared, _ = strip_tool_specific(repo.text(router))
        prose = FENCE.sub("\n", shared)
        # Routers reference knowledge two ways: as markdown links and as
        # backticked paths. Both are references, so both count - a router that
        # names `rules/README.md` in backticks while another links to it is not
        # forked, and one that omits it entirely is.
        referenced = {
            _resolve(repo, router, t) for t in MD_LINK.findall(prose)
            if not _is_external(t)
        }
        referenced |= {t.replace("\\", "/") for t in BACKTICK_PATH.findall(prose)}
        knowledge_links[rel] = {
            t for t in referenced
            if t.split("/", 1)[0] in KNOWLEDGE_DIRS
            and "*" not in t          # a glob describes a set, not an artifact
            and repo.exists_rel(t)    # a broken reference is a dead-links finding
        }

    union: set[str] = set()
    for targets in knowledge_links.values():
        union |= targets

    findings: list[Finding] = []
    for rel, targets in sorted(knowledge_links.items()):
        missing = sorted(union - targets)
        if not missing:
            continue
        findings.append(Finding(
            check="router-sync",
            severity="high",
            path=rel,
            message="router fork - other routers reference knowledge this one "
                    f"omits: {', '.join(missing[:6])}"
                    + (f" (+{len(missing) - 6} more)" if len(missing) > 6 else ""),
            fix="Update every router in the same change, or mark the intentional "
                "difference with <!-- akinator:tool-specific --> (skill: "
                "akinator-router-sync).",
        ))
    return findings


def check_module_routers(repo: Repo) -> list[Finding]:
    if not repo.has_knowledge_layer():
        return []
    findings: list[Finding] = []
    for module in repo.modules():
        if any((module / name).exists() for name in ROUTER_NAMES):
            continue
        findings.append(Finding(
            check="module-routers",
            severity="medium",
            path=repo.rel(module),
            message="module or service has no local router file",
            fix="Add <module>/CLAUDE.md stating this module's local rules, entry "
                "points, tests and gotchas, linked up to the root router "
                "(skill: akinator-router-sync).",
        ))
    return findings


def check_generated(repo: Repo) -> list[Finding]:
    findings: list[Finding] = []
    for path in repo.markdown:
        # Templates and their filled examples show what a generated artifact
        # looks like; they are not generated artifacts of this repository. A
        # filled example necessarily names a generator that exists only in the
        # fiction it illustrates.
        if repo.rel(path).startswith("templates/"):
            continue

        # A generated-file banner is an HTML comment at the top. Scanning any
        # prose in the first 600 characters was wrong: a ledger record whose
        # subject was "a generated file multiplies one stale reference" tripped
        # the check by describing the thing rather than being it. False
        # positives are how a checker loses its reader.
        banner_match = LEADING_COMMENT.match(repo.text(path))
        if not banner_match:
            continue
        head = banner_match.group(0).lower()
        if not any(marker in head for marker in GENERATED_MARKERS):
            continue
        rel = repo.rel(path)
        banner_region = banner_match.group(0)

        # A banner naming a placeholder generator - `scripts/<extractor>.py` -
        # is a template showing what a generated file looks like, not a claim
        # that this file was generated.
        if re.search(r"`[^`]*<[^`>]+>[^`]*`", banner_region):
            continue

        # A generator may be named by path (`scripts/gen.py`) or by bare
        # filename. The bare form matters for a file a repository both generates
        # and keeps: the name resolves without asserting a layout.
        named = BACKTICK_PATH.findall(banner_region)
        generators = [
            t for t in named
            if repo.exists_rel(t.replace("\\", "/"))
            or ("/" not in t and any(p.name == t for p in repo.files))
        ]
        if generators:
            continue

        # Named-but-absent is checked BEFORE the vendored branch below, and
        # deliberately so. A banner that still points at a generator is making a
        # claim about a local file, and that claim is checked whatever else the
        # banner says - otherwise "installed from" becomes a phrase you write to
        # silence the check, which is the gaming pattern this repo rates worst.
        if named:
            findings.append(Finding(
                check="generated",
                severity="high",
                path=rel,
                message="declares itself generated but its named generator does "
                        f"not exist: {', '.join(sorted(set(named))[:3])}",
                fix="Point the banner at the real extractor, or stop declaring "
                    "the file generated (skill: akinator-contextify).",
            ))
            continue

        # Nothing is named. That is correct for a **vendored** artifact: it was
        # installed from somewhere else and the generator is deliberately not in
        # this tree, so demanding one inverts the check and turns a correct file
        # into a finding. This was not hypothetical - installing Akinator's own
        # Codex pack produced 22 HIGH findings in the target repository on the
        # first run, for exactly this reason.
        #
        # What such a file owes its reader is not a local script but its origin
        # and a way to refresh it. Both halves are required: an origin with no
        # refresh path is a dead end, and a stale copy that cannot be updated is
        # no better than a generator that is not there.
        if VENDORED_ORIGIN.search(banner_region):
            if VENDORED_UPDATE.search(banner_region):
                continue
            findings.append(Finding(
                check="generated",
                severity="high",
                path=rel,
                message="declares itself installed from elsewhere but gives no "
                        "way to update it",
                fix="State how to refresh the vendored copy - normally by "
                    "reinstalling whatever placed it here - so a stale copy is "
                    "fixable rather than merely unexplained.",
            ))
            continue

        findings.append(Finding(
            check="generated",
            severity="medium",
            path=rel,
            message="declares itself generated but names no generator",
            fix="Name the extractor in the banner, by path, in backticks - or, "
                "if the file is vendored, say where it was installed from and "
                "how to refresh it.",
        ))
    return findings


def check_doc_truth(repo: Repo, sample: int) -> list[Finding]:
    findings: list[Finding] = []
    docs = sorted(set(repo.knowledge_artifacts()) | set(repo.skills()))
    for doc in docs:
        rel = repo.rel(doc)
        broken: list[str] = []
        for token in path_mentions(repo, doc, sample):
            # Only judge tokens that look like in-repo paths: they must contain a
            # separator, or match a known top-level knowledge file.
            if "/" not in token:
                continue
            if token.startswith(("http", "$", "<", "@")):
                continue
            if repo.exists_rel(token):
                continue
            # Tolerate illustrative placeholders.
            if any(ch in token for ch in ("*", "<", ">")) or "NN" in token:
                continue
            if re.search(r"\b(example|template|placeholder|foo|bar)\b", token, re.I):
                continue
            broken.append(token)
        if broken:
            findings.append(Finding(
                check="doc-truth",
                severity="high",
                path=rel,
                message="names paths that do not exist in the tree: "
                        + ", ".join(sorted(set(broken))[:6]),
                fix="Correct or remove them. A doc describing what is not there "
                    "is trusted and wrong (skill: akinator-document-change).",
            ))
    return findings


REQUIRED_SKILL_SECTIONS = (
    ("when to use", "high"),
    ("when not to use", "medium"),
    ("procedure", "high"),
    ("definition of done", "high"),
)


def check_skill_format(repo: Repo) -> list[Finding]:
    findings: list[Finding] = []
    for skill in repo.skills():
        rel = repo.rel(skill)
        text = repo.text(skill)
        if not text.startswith("---"):
            findings.append(Finding(
                check="skill-format",
                severity="critical",
                path=rel,
                message="has no YAML frontmatter - the skill cannot be "
                        "discovered or triggered",
                fix="Add frontmatter with name and description (skill: "
                    "akinator-skillify).",
            ))
            continue
        end = text.find("\n---", 3)
        frontmatter = text[3:end] if end != -1 else ""
        for key in ("name:", "description:"):
            if key not in frontmatter:
                findings.append(Finding(
                    check="skill-format",
                    severity="critical",
                    path=rel,
                    message=f"frontmatter is missing '{key.rstrip(':')}'",
                    fix="A skill without a trigger description never fires "
                        "(skill: akinator-skillify).",
                ))
        lowered = text.lower()
        for section, severity in REQUIRED_SKILL_SECTIONS:
            if section not in lowered:
                findings.append(Finding(
                    check="skill-format",
                    severity=severity,
                    path=rel,
                    message=f"missing the '{section}' section",
                    fix="Every skill carries when-to-use, when-NOT-to-use, "
                        "procedure, failure modes and definition of done "
                        "(skill: akinator-skillify).",
                ))
    return findings


STALENESS_PHRASES = (
    "regenerate when", "review when", "stale when", "revisit when",
    "what would make this stale", "last verified", "do not edit",
    "auto-generated", "generated by",
)


def check_staleness(repo: Repo) -> list[Finding]:
    base = repo.root / "context"
    if not base.is_dir():
        return []
    findings: list[Finding] = []
    for path in repo.markdown:
        if base not in path.parents and path.parent != base:
            continue
        if path.name.lower() in ("readme.md", "index.md"):
            continue
        lowered = repo.text(path).lower()
        if any(phrase in lowered for phrase in STALENESS_PHRASES):
            continue
        findings.append(Finding(
            check="staleness",
            severity="medium",
            path=repo.rel(path),
            message="context map states no regenerate-or-review trigger",
            fix="Add 'Regenerate when' (generated) or 'Review when' plus a "
                "last-verified date (manual) (skill: akinator-contextify).",
        ))
    return findings


KNOWLEDGE_WORDS = (
    "doc", "docs", "documentation", "knowledge", "coverage", "akinator",
    "rules/", "context/", "memory/", "adr",
)


def check_git_hooks(repo: Repo) -> list[Finding]:
    """A knowledge check wired into a git hook is a hard design violation."""
    candidates: list[Path] = []
    hooks_dir = repo.root / ".git" / "hooks"
    if hooks_dir.is_dir():
        candidates += [p for p in hooks_dir.iterdir()
                       if p.is_file() and not p.name.endswith(".sample")]
    husky = repo.root / ".husky"
    if husky.is_dir():
        candidates += [p for p in husky.iterdir() if p.is_file()]
    precommit = repo.root / ".pre-commit-config.yaml"
    if precommit.exists():
        candidates.append(precommit)

    findings: list[Finding] = []
    for path in candidates:
        try:
            text = path.read_text(encoding="utf-8", errors="replace").lower()
        except OSError:
            continue
        hits = sorted({w for w in KNOWLEDGE_WORDS if w in text})
        if not hits:
            continue
        findings.append(Finding(
            check="git-hooks",
            severity="critical",
            path=repo.rel(path),
            message="a git hook appears to run knowledge or documentation "
                    f"checks (matched: {', '.join(hits[:4])})",
            fix="Remove it. Git hooks gate code and must stay fast; knowledge "
                "enforcement belongs in CI, the test suite and session "
                "behavior (rules/05-no-git-hook-complication.md).",
        ))
    return findings


# --------------------------------------------------------------------------
# Runner
# --------------------------------------------------------------------------

def run_checks(repo: Repo, only: Sequence[str], skip: Sequence[str],
               sample: int) -> list[Finding]:
    registry = {
        "reachability": lambda: check_reachability(repo),
        "index-completeness": lambda: check_index_completeness(repo),
        "dead-links": lambda: check_dead_links(repo),
        "rule-enforcement": lambda: check_rule_enforcement(repo),
        "router-sync": lambda: check_router_sync(repo),
        "module-routers": lambda: check_module_routers(repo),
        "generated": lambda: check_generated(repo),
        "doc-truth": lambda: check_doc_truth(repo, sample),
        "skill-format": lambda: check_skill_format(repo),
        "staleness": lambda: check_staleness(repo),
        "git-hooks": lambda: check_git_hooks(repo),
    }
    selected = [name for name in registry
                if (not only or name in only) and name not in skip]
    findings: list[Finding] = []
    for name in selected:
        findings.extend(registry[name]())
    findings.sort(key=Finding.sort_key)
    return findings


def report(repo: Repo, findings: list[Finding], threshold: str) -> str:
    lines: list[str] = []
    lines.append("Akinator coverage report")
    lines.append(f"repo: {repo.root}")
    lines.append("")

    if not repo.has_knowledge_layer():
        lines.append("No knowledge layer detected.")
        lines.append("This repo has no routers and none of "
                     f"{', '.join(KNOWLEDGE_DIRS)}.")
        lines.append("Onboard it first: ask your agent to onboard this "
                     "repository - Akinator is always on - or run "
                     "/akinator:everything.")
        lines.append("")

    counts = {level: 0 for level in SEVERITIES}
    for finding in findings:
        counts[finding.severity] += 1

    lines.append("  ".join(
        f"{level}: {counts[level]}" for level in SEVERITIES
    ))
    lines.append("")

    if not findings:
        lines.append("All invariants hold.")
        return "\n".join(lines)

    current = None
    for finding in findings:
        if finding.severity != current:
            current = finding.severity
            lines.append(f"--- {current.upper()} ---")
        lines.append(f"[{finding.check}] {finding.path}")
        lines.append(f"    {finding.message}")
        if finding.fix:
            lines.append(f"    fix: {finding.fix}")
        lines.append("")

    blocking = [f for f in findings if _SEV_RANK[f.severity] <= _SEV_RANK[threshold]]
    lines.append(
        f"{len(blocking)} finding(s) at or above '{threshold}' - "
        + ("FAIL" if blocking else "PASS")
    )
    return "\n".join(lines)


def main(argv: Sequence[str] | None = None) -> int:
    parser = argparse.ArgumentParser(
        prog="akinator_coverage",
        description="Verify the Akinator knowledge invariants against a repo.",
    )
    parser.add_argument("root", nargs="?", default=".",
                        help="repository root (default: current directory)")
    parser.add_argument("--json", action="store_true", dest="as_json")
    parser.add_argument("--strict", action="store_true",
                        help="fail on MEDIUM findings too")
    parser.add_argument("--fail-on", default="high", choices=SEVERITIES)
    parser.add_argument("--only", default="")
    parser.add_argument("--skip", default="")
    parser.add_argument("--sample", type=int, default=40)
    parser.add_argument(
        "--exclude", action="append", default=[],
        help="path prefix or glob to skip, in addition to .akinatorignore; "
             "repeatable",
    )
    parser.add_argument("--list-checks", action="store_true")
    args = parser.parse_args(argv)

    if args.list_checks:
        for name, description in CHECKS.items():
            print(f"{name:18} {description}")
        return 0

    root = Path(args.root).resolve()
    if not root.is_dir():
        print(f"not a directory: {root}", file=sys.stderr)
        return 2

    only = [s.strip() for s in args.only.split(",") if s.strip()]
    skip = [s.strip() for s in args.skip.split(",") if s.strip()]
    unknown = [name for name in only + skip if name not in CHECKS]
    if unknown:
        print(f"unknown check(s): {', '.join(unknown)}", file=sys.stderr)
        return 2

    threshold = "medium" if args.strict else args.fail_on

    repo = Repo(root, args.exclude)
    findings = run_checks(repo, only, skip, args.sample)

    if args.as_json:
        payload = {
            "root": str(root),
            "threshold": threshold,
            "counts": {
                level: sum(1 for f in findings if f.severity == level)
                for level in SEVERITIES
            },
            "findings": [asdict(f) for f in findings],
        }
        print(json.dumps(payload, indent=2, sort_keys=True))
    else:
        print(report(repo, findings, threshold))

    blocking = [f for f in findings
                if _SEV_RANK[f.severity] <= _SEV_RANK[threshold]]
    return 1 if blocking else 0


if __name__ == "__main__":
    raise SystemExit(main())

SHA-256: dd27b66b82be077605ae32bbc3c429acf579a2548e524547e2dfc89adb0db60c