← Files taskplaneARCHIVED FILE

taskplane/graph_primitives.py

52.2 KB · Oct 2, 2026 · 00:29 UTC

↓ Download file

"""Lower-owned graph identities, context records, and lens applicability.

This module is the single dependency-free contract shared by the graph
scanner, component decomposition, and lens facade.  It owns the deterministic
component lens engine so direct graph APIs need no higher-layer activation.
"""
from __future__ import annotations

import json
import os
import posixpath
import re

if __package__:
    from . import glob_match, path_roles
else:
    import glob_match
    import path_roles

_is_test_path = path_roles.is_test_path
_change_adds_no_test = path_roles.change_adds_no_test


# Directory names that mark source layout rather than component identity.
_SRC_ROOTS = ("src", "app", "lib", "packages", "pkg", "internal", "cmd")
_GO_MODULE_LINE = re.compile(r"^\s*module\s+(\S+)", re.M)
_ID_PREFIXES = ("ext:", "svc:", "req:", "contract:", "resource:")

# Reserved manifest-map key for the repository's own Go module path.  The
# leading NUL cannot collide with a real repository directory.
ROOT_MODULE_KEY = "\x00root_module"

# Edges whose direction means "from NEEDS to".  Structural and annotation
# edges deliberately do not answer dependent/blast-radius questions.
DEPENDENCY_EDGE_KINDS = frozenset({
    "imports", "depends_on", "consumes", "depends", "calls", "uses",
})

_FIXTURE_SEGMENTS = frozenset({"fixtures", "testdata", "goldens"})
_GRAPH_LOADER = None


def register_graph_loader(loader) -> None:
    """Register the scanner-owned read boundary used by lens projection."""
    global _GRAPH_LOADER
    _GRAPH_LOADER = loader


def load_graph(workspace: str) -> dict:
    """Read through the registered persistence boundary.

    Persistence remains owned by depgraph; this lower layer only holds the
    injected callable, which keeps lens routing from importing the scanner.
    """
    if _GRAPH_LOADER is None:
        raise RuntimeError("graph loader is not registered")
    return _GRAPH_LOADER(workspace)


def is_fixture_module(module: str) -> bool:
    """Whether a module boundary itself is fixture-classed."""
    value = str(module or "").replace("\\", "/").strip("/")
    if not value or value == "(root)":
        return False
    return (any(part.lower() in _FIXTURE_SEGMENTS
                for part in value.split("/") if part)
            or value.lower().endswith(".golden"))


def manifest_modules(files, read) -> dict:
    """Return ``{directory: declared import identity}`` for owned manifests.

    Only package.json names and go.mod module paths are import identities.
    Root package manifests describe the repository, while a root go.mod path
    is retained solely as an import prefix under ``ROOT_MODULE_KEY``.
    """
    out: dict = {}
    for rel in sorted(files or ()):
        rel = str(rel).replace("\\", "/")
        base = posixpath.basename(rel)
        if base not in ("package.json", "go.mod"):
            continue
        directory = posixpath.dirname(rel)
        if not directory:
            if base == "go.mod":
                match = _GO_MODULE_LINE.search(read(rel) or "")
                if match and match.group(1) and "/" in match.group(1):
                    out[ROOT_MODULE_KEY] = match.group(1).strip()
            continue
        text = read(rel)
        if not text:
            continue
        declared = None
        if base == "package.json":
            try:
                data = json.loads(text)
            except (ValueError, TypeError):
                continue
            if isinstance(data, dict) and isinstance(data.get("name"), str):
                declared = data["name"].strip()
        else:
            match = _GO_MODULE_LINE.search(text)
            declared = match.group(1).strip() if match else None
        if not declared or declared.startswith(_ID_PREFIXES):
            continue
        out[directory] = declared.replace("\\", "/").strip("/")
    return {key: value for key, value in out.items() if value}


def declared_module_ids(graph: dict | None) -> dict:
    """Return the manifest identity map persisted by the last graph scan."""
    return ((graph or {}).get("meta") or {}).get("module_ids") or {}


def root_module(declared_ids) -> "str | None":
    """Return the repository Go module prefix when held in map form."""
    if isinstance(declared_ids, dict):
        return declared_ids.get(ROOT_MODULE_KEY) or None
    return None


def strip_root_prefix(spec: str, root: "str | None") -> "str | None":
    """Map ``<root>/pkg/x`` to the repository-relative ``pkg/x``."""
    if not root:
        return None
    spec = str(spec or "").replace("\\", "/").strip("/")
    if spec == root:
        return None
    if spec.startswith(root + "/"):
        return spec[len(root) + 1:] or None
    return None


def _strip_root_module(spec: str, declared_ids) -> "str | None":
    return strip_root_prefix(spec, root_module(declared_ids))


def _declared_target(spec: str, declared_ids) -> "str | None":
    """Return the longest declared module matching an import specifier."""
    if not declared_ids:
        return None
    spec = str(spec or "").replace("\\", "/").strip("/")
    while spec:
        if spec in declared_ids and spec != ROOT_MODULE_KEY:
            return spec
        if "/" not in spec:
            return None
        spec = spec.rsplit("/", 1)[0]
    return None


def module_of(relpath: str, manifests: dict | None = None) -> str:
    """Return the stable module identity owning a repository-relative path."""
    relpath = str(relpath or "").replace("\\", "/")
    directory = posixpath.dirname(relpath)
    if manifests:
        probe = directory
        while probe:
            hit = manifests.get(probe)
            if hit:
                return hit
            parent = posixpath.dirname(probe)
            if parent == probe:
                break
            probe = parent
    if not directory:
        return "(root)"
    parts = directory.split("/")
    for marker in (("src", "main", "java"), ("src", "test", "java")):
        for index in range(0, len(parts) - len(marker) + 1):
            if tuple(parts[index:index + len(marker)]) == marker:
                package = parts[index + len(marker):]
                if package:
                    return "/".join(package[-3:])
    kept = [part for part in parts if part not in _SRC_ROOTS]
    return "/".join(kept[:2]) if kept else parts[-1]


def node_kind(node: str) -> str:
    """Return the public graph-node family for an identifier."""
    if node.startswith("component:"):
        return "component"
    if node.startswith("surface:"):
        return "surface"
    if node.startswith("req:"):
        return "requirement"
    if node.startswith("contract:"):
        return "contract"
    if node.startswith("resource:"):
        return "resource"
    if node.startswith("svc:"):
        return "infra"
    if node.startswith("ext:"):
        return "external"
    return "module"


def is_boundary(node: str) -> bool:
    return node.startswith(("contract:", "resource:", "svc:", "ext:"))


def is_dependency_edge(edge: dict) -> bool:
    """Whether ``edge`` expresses the dependency direction from NEEDS to."""
    try:
        return edge.get("kind") in DEPENDENCY_EDGE_KINDS
    except AttributeError:
        return False


def strongly_connected_components(nodes, edges) -> list[list[str]]:
    """Return a deterministic, complete Tarjan decomposition.

    ``edges`` contains ``(source, target)`` pairs. Unknown endpoints are not
    silently admitted: the caller must validate them before asking for SCC
    evidence.  This helper never truncates; bounded callers must refuse input
    over their declared ceiling before calling it.
    """
    ordered = sorted({str(node) for node in nodes or ()})
    known = set(ordered)
    adjacency = {node: set() for node in ordered}
    for source, target in edges or ():
        source, target = str(source), str(target)
        if source not in known or target not in known:
            raise ValueError(
                f"SCC edge names an unknown node: {source} -> {target}")
        adjacency[source].add(target)

    index = 0
    indexes: dict[str, int] = {}
    lowlinks: dict[str, int] = {}
    stack: list[str] = []
    on_stack: set[str] = set()
    components: list[list[str]] = []

    def visit(node: str) -> None:
        nonlocal index
        indexes[node] = lowlinks[node] = index
        index += 1
        stack.append(node)
        on_stack.add(node)
        for target in sorted(adjacency[node]):
            if target not in indexes:
                visit(target)
                lowlinks[node] = min(lowlinks[node], lowlinks[target])
            elif target in on_stack:
                lowlinks[node] = min(lowlinks[node], indexes[target])
        if lowlinks[node] != indexes[node]:
            return
        component = []
        while stack:
            member = stack.pop()
            on_stack.remove(member)
            component.append(member)
            if member == node:
                break
        components.append(sorted(component))

    for node in ordered:
        if node not in indexes:
            visit(node)
    return sorted(components, key=lambda component: tuple(component))


def graph_payload(graph: dict, modules,
                  *, fixture_module_predicate=None) -> dict:
    """Project raw graph data into the shared lens/component context record.

    ``hub_dependents`` intentionally counts every incoming edge for backward
    compatibility. ``module_dependents`` is stricter: only dependency edges,
    excluding self-dependence and fixture-class witnesses.  The result stays
    an ordinary dict so all existing JSON and caller payloads remain stable.
    """
    selected = sorted({str(module) for module in modules or () if module})
    selected_set = set(selected)
    incoming: dict[str, set] = {}
    dependency_incoming: dict[str, set] = {}
    contracts: set[str] = set()
    fixture_module_predicate = (
        fixture_module_predicate or (lambda _value: False))

    for edge in (graph or {}).get("edges") or []:
        if not isinstance(edge, dict):
            continue
        source, target = edge.get("from"), edge.get("to")
        if not source or not target:
            continue
        incoming.setdefault(target, set()).add(source)
        if is_dependency_edge(edge):
            dependency_incoming.setdefault(target, set()).add(source)
        for left, right in ((source, target), (target, source)):
            if left in selected_set and str(right).startswith("contract:"):
                contracts.add(str(right))

    return {
        "hub_dependents": max(
            (len(incoming.get(module, ())) for module in selected), default=0),
        "boundary_contracts": sorted(contracts),
        "modules": selected,
        "module_ids": declared_module_ids(graph),
        "module_dependents": {
            module: len([
                dependent
                for dependent in dependency_incoming.get(module, ())
                if dependent != module
                and not fixture_module_predicate(str(dependent))
            ])
            for module in selected
        },
    }


# ---------------------------------------------------------------- thresholds

DEEP = 0.6            # score >= DEEP  -> "deep"
LIGHT = 0.2           # score >= LIGHT -> "light"; below -> "n/a"

MAX_FILE_BYTES = 64 * 1024   # per-file content-scan bound
MAX_FILES = 200              # max files content-scanned per ctx

# signal weights (sum is clamped to 1.0)
W_PATH = 0.35      # a changed path matches the lens's surface globs
W_CONTENT = 0.3    # a content marker fires in a changed file
W_DENSITY = 0.25   # density-style content signal (e.g. user-facing strings)
W_KEYWORD = 0.15   # the requirement text mentions the lens's concern
W_GRAPH = 0.35     # graph flag (hub module / boundary contract in impact)

_HUB_DEPENDENTS = 3   # >= this many direct dependents -> hub signal fires

# ------------------------------------------- fixtures-path discount (D-0002)
#
# Checked-in test fixtures LOOK like the surfaces they imitate (a locale
# file under tests/fixtures/ scores like a real locale file), so fixture-only
# diffs inflated i18n/mobile to deep (D-0002). Path/content/density signal
# hits whose ONLY support is fixture-class files are RE-WEIGHTED x0.25 —
# never suppressed: the evidence line survives and names the discount
# (honesty: the evidence says why the score is low), n/a semantics are
# untouched, and the floors still apply after scoring. A hit with any real
# product-file support keeps full weight.

FIXTURE_DISCOUNT = 0.25
_FIXTURE_SEGMENTS = frozenset({"fixtures", "testdata", "goldens"})
_FIXTURE_EXTENSIONS = (".golden",)
_DISCOUNT_NOTE = f"(fixture-path discount x{FIXTURE_DISCOUNT})"


def is_fixture_path(path: str) -> bool:
    """True when the path is fixture-class: any path segment named
    fixtures/testdata/goldens (at any depth), or a .golden extension."""
    p = str(path).replace(os.sep, "/")
    if any(seg.lower() in _FIXTURE_SEGMENTS for seg in p.split("/") if seg):
        return True
    return p.lower().endswith(_FIXTURE_EXTENSIONS)


# B5 (R-0008): the classifier above keys off the PATH alone, so a REAL
# product directory literally named `fixtures/` was discounted — the
# dangerous direction (under-routing). The discount APPLICATION point now
# consults a graph-informed exemption computed at ctx construction: a
# fixture-classed path whose containing graph MODULE has at least
# FIXTURE_EXEMPT_MIN_DEPENDENTS dependents is real product code (nothing
# depends on a test-fixture tree), keeps FULL weight, and says so in the
# evidence line. The exemption can only ever RESTORE weight — the
# discounted set strictly shrinks, never grows — and is_fixture_path()
# itself is untouched, so every other caller sees the same classification.
#
# TWO guards keep that premise TRUE, because the first cut of this exemption
# repealed D-0002 on this very repository:
#
#  1. WHAT COUNTS AS A DEPENDENT — only depgraph.DEPENDENCY_EDGE_KINDS
#     ("from NEEDS to"). `module_dependents` used to count EVERY incoming
#     edge, including the STRUCTURAL `defined_in` edge a docker-compose file
#     emits ABOUT ITS OWN module. This repo's
#     taskplane/tests/fixtures/detectors/architecture/positive/
#     docker-compose.yml therefore handed module `taskplane/tests` two
#     "dependents" that were the fixtures themselves. Filtered in
#     _graph_payload here AND in decompose._graph_payload, so cached
#     component maps and live routing agree.
#
#  2. AT WHAT GRANULARITY — a graph module id is at most a few segments
#     (`taskplane/tests`) while is_fixture_path classifies the FULL path
#     (`taskplane/tests/fixtures/detectors/i18n/positive/locales/en.json`).
#     An ancestor module id must never speak for a fixture subtree nested
#     below it, so the exemption fires only when the fixture classification
#     is visible AT the module boundary that carries the dependent count —
#     is_fixture_path(module) is itself True (`fixtures`, `tests/fixtures`,
#     …). When the fixture-marking segment lies BELOW the module id, the
#     dependents belong to the module, not to the fixture tree, and the
#     discount stands. On this repo all 102 fixture-classed tracked paths
#     map to non-fixture-classed modules (taskplane/tests, src, api,
#     components, auth), so the whole corpus stays discounted.
#
# Both guards only SHRINK the exemption, and the exemption only ever restores
# weight, so behaviour stays between "always discounted" (base) and the
# unguarded version: it can never route anything more narrowly than base.
FIXTURE_EXEMPT_MIN_DEPENDENTS = 1


def _module_is_fixture_classed(module: str) -> bool:
    """Guard 2: the fixture marking must be visible at the module boundary
    that carries the dependent count, not somewhere deeper in the path."""
    return is_fixture_module(module)


def _module_of(path: str, manifests: dict | None = None) -> str:
    """The graph module id owning ``path`` through the shared contract."""
    p = str(path).replace(os.sep, "/")
    try:
        return module_of(p, manifests)
    except Exception:
        d = os.path.dirname(p)
        return "/".join(d.split("/")[:2]) if d else "(root)"


def _fixture_exemptions(files, graph) -> dict:
    """{path: reason} — the fixture-classed paths that keep FULL weight.

    Guard 2 (granularity) is applied here; guard 1 (dependency-edge kinds)
    is applied where `module_dependents` is BUILT (_graph_payload)."""
    deps = (graph or {}).get("module_dependents")
    if not isinstance(deps, dict) or not deps:
        return {}
    ids = (graph or {}).get("module_ids") or None
    out: dict = {}
    for rel in files:
        if not is_fixture_path(rel):
            continue
        module = _module_of(rel, ids)
        if not _module_is_fixture_classed(module):
            # the fixture segment sits BELOW the module boundary: this
            # module's dependents do not describe the fixture tree
            continue
        try:
            n = int(deps.get(module, 0) or 0)
        except (TypeError, ValueError):
            n = 0
        if n >= FIXTURE_EXEMPT_MIN_DEPENDENTS:
            out[rel] = (f"product-dir exemption: {rel} is in module "
                        f"{module}, which has {n} dependent(s) — real "
                        "product code, not a test fixture")
    return out


# ------------------------------------------------------------------- catalog

_CATALOG_CACHE: dict = {}


def _plugin_root() -> str:
    # lenses/ sits at the plugin root, one level up from taskplane/
    return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))


def load_catalog(root: str | None = None) -> dict:
    """Load lenses/catalog.json (cached per root). Self-contained on purpose:
    lens.py will import THIS module in t2, so importing lens here would set
    up an import cycle."""
    key = root or _plugin_root()
    if key not in _CATALOG_CACHE:
        with open(os.path.join(key, "lenses", "catalog.json"), encoding="utf-8") as f:
            _CATALOG_CACHE[key] = json.load(f)
    return _CATALOG_CACHE[key]


# -------------------------------------------------------------- glob matcher

def _match(path: str, glob: str) -> bool:
    """Compatibility facade over the shared dependency-neutral matcher."""
    return glob_match.path_matches(path, glob)


def _glob_hit(files, globs):
    """First (file, glob) pair that matches, or None. files/globs iterated
    in the given (already sorted) order -> deterministic."""
    return glob_match.first_match(files, globs)


def _is_code(path: str, code_ext) -> bool:
    return any(path.endswith(e) for e in code_ext)


# ------------------------------------------------------------------- context

class Ctx:
    """Everything a detector may look at. Bounded, cached, deterministic.

    files            changed paths relative to workspace (sorted, deduped)
    workspace        absolute-ish root used to read file contents
    requirement_text lowercased requirement/acceptance-criteria blob
    graph            {"hub_dependents": int, "boundary_contracts": [str],
                      "modules": [str], "module_dependents": {mod: int}}
    stage            delivery stage (context only; unused by detectors)
    fixture_exempt   {path: reason} — fixture-classed paths that keep FULL
                     weight because their graph module has dependents (B5,
                     R-0008); computed HERE, at ctx construction
    """

    __slots__ = ("workspace", "files", "requirement_text", "graph", "stage",
                 "fixture_exempt", "_contents", "_content_by_file")

    def __init__(self, workspace, files, requirement_text, graph, stage,
                 content_by_file=None):
        self.workspace = workspace
        self.files = sorted({str(f).replace(os.sep, "/") for f in files or []})
        if isinstance(requirement_text, (list, tuple)):
            requirement_text = "\n".join(str(x) for x in requirement_text)
        self.requirement_text = (requirement_text or "").lower()
        self.graph = graph or {"hub_dependents": 0,
                               "boundary_contracts": [], "modules": []}
        self.stage = stage
        self.fixture_exempt = _fixture_exemptions(self.files, self.graph)
        self._content_by_file = (
            {str(path).replace(os.sep, "/"): str(text)
             for path, text in content_by_file.items()
             if str(path).replace(os.sep, "/") in self.files}
            if isinstance(content_by_file, dict) else None)
        self._contents = None

    def is_discounted(self, path: str) -> bool:
        """True when the D-0002 fixture discount APPLIES to `path`: it is
        fixture-class AND not graph-exempt (B5). A strict subset of
        is_fixture_path() — the exemption can only restore weight."""
        p = str(path).replace(os.sep, "/")
        return is_fixture_path(p) and p not in self.fixture_exempt

    def exemption_note(self, path: str) -> str:
        """' (reason)' when `path` is exempt from the discount, else ''."""
        reason = self.fixture_exempt.get(str(path).replace(os.sep, "/"))
        return f" ({reason})" if reason else ""

    def read(self, relpath: str) -> str | None:
        """Bounded read of one changed file: at most MAX_FILE_BYTES bytes,
        decoded utf-8 with replacement. Missing/unreadable -> None (changed
        lists legitimately contain deletions)."""
        p = os.path.join(self.workspace, relpath)
        try:
            # Containment (EM v3): changed-file lists come from git, but a
            # crafted relpath ('../..') or a symlink pointing outside the
            # workspace must not let a detector read foreign files. Resolve
            # and require the real target stays under the real workspace.
            root = os.path.realpath(self.workspace)
            real = os.path.realpath(p)
            if real != root and not real.startswith(root + os.sep):
                return None
            with open(real, "rb") as f:
                text = f.read(MAX_FILE_BYTES).decode("utf-8", "replace")
            # Read as bytes (deliberately — no locale codec, no newline
            # translation), then normalize line endings ourselves. Detector
            # regexes are line-anchored and the scores they produce are
            # frozen in goldens; a file checked out with CRLF must score
            # identically to the same file with LF, or the same diff routes
            # differently on Windows than it does in CI.
            return text.replace("\r\n", "\n").replace("\r", "\n")
        except OSError:
            return None

    def contents(self):
        """[(relpath, text)] for the first MAX_FILES sorted changed files
        that exist. Cached: the corpus is read once per ctx, then every
        detector scans the same in-memory snapshot."""
        if self._contents is None:
            out = []
            for rel in self.files:
                if len(out) >= MAX_FILES:
                    break
                if self._content_by_file is not None:
                    text = self._content_by_file.get(rel)
                    if text is not None:
                        text = text.encode("utf-8")[:MAX_FILE_BYTES].decode(
                            "utf-8", "replace")
                else:
                    text = self.read(rel)
                if text is not None:
                    out.append((rel, text))
            self._contents = out
        return self._contents


def make_ctx(workspace, files, requirement_text=None, graph=None,
             stage=None, content_by_file=None) -> Ctx:
    """Build source-signal context; unavailable graphs yield an empty payload."""
    if graph is None:
        graph = _graph_payload(workspace, files)
    return Ctx(workspace, files, requirement_text, graph, stage,
               content_by_file=content_by_file)


def _graph_payload(workspace, files) -> dict:
    try:
        g = load_graph(workspace)
        # the graph's own declared ids, or the fixture-exemption and
        # dependent-count lookups below miss every workspace module
        _ids = declared_module_ids(g)
        touched = sorted({module_of(
                              str(f).replace(os.sep, "/"), _ids)
                          for f in files or []})
        return graph_payload(
            g, touched,
            fixture_module_predicate=is_fixture_module)
    except Exception:
        return {"hub_dependents": 0, "boundary_contracts": [], "modules": [],
                "module_dependents": {}}


# --------------------------------------------------------- signal-spec table
#
# One spec per catalog lens. Shape:
#   paths:    extra surface globs ON TOP of the lens's catalog globs
#             (the catalog globs are always part of the path signal)
#   code:     True -> the path signal also fires on any code-extension file
#             (baseline lenses: their surface IS "code changed")
#   content:  [(label, regex_pattern)] scanned over the bounded corpus;
#             each distinct rule that fires adds W_CONTENT
#   density:  (label, threshold) -> user-facing string-literal density rule
#   graph:    subset of {"hub", "boundary"}
#   keywords: substrings looked up in the lowercased requirement text
#   absent:   negative-evidence phrases, joined into
#             "0 <lens> signals: no X, no Y, ..." when nothing fires

_STR_LIT = re.compile(r"""["']([A-Za-z][A-Za-z,.!?'-]*(?:\s+[A-Za-z][A-Za-z,.!?'-]*){2,})["']""")

SPECS: dict[str, dict] = {
    "product": {
        "content": [("spec/acceptance markers",
                     r"(?im)^#+.*(acceptance criteria|user stor|requirement)")],
        "keywords": ["user journey", "acceptance", "success metric",
                     "user value"],
        "absent": ["no spec/requirements files", "no acceptance-criteria "
                   "markers", "no product keywords in the requirement"],
    },
    "security": {
        "paths": ["**/hooks/**", "**/taskplane_lite.py", "**/*login*",
                  "**/*permission*", "**/*.pem", "**/*credential*"],
        "content": [
            ("auth/secret markers",
             r"(?i)(password|passwd|secret[_a-z]*\s*=|api[_-]?key|"
             r"bearer\s|jwt|oauth|csrf|bcrypt|hmac|authenticat|authoriz|"
             r"permission|session[_ ]token)"),
            ("unsafe-input surface",
             r"(?i)(\beval\(|\bexec\(|subprocess|os\.system|pickle\.loads|"
             r"yaml\.load\(|innerHTML|dangerouslySetInnerHTML|"
             r"shell\s*=\s*True)"),
        ],
        "graph": ["boundary"],
        "keywords": ["auth", "security", "secret", "permission", "vulnerab",
                     "enforc", "injection"],
        "absent": ["no auth/secrets/enforcement paths", "no auth or secret "
                   "markers", "no unsafe-input surface", "no boundary "
                   "contracts in impact", "no security keywords in the "
                   "requirement"],
    },
    "code-quality": {
        "code": True,
        "content": [("code constructs",
                     r"(?m)^\s*(def |class |function\b|const |public |"
                     r"private |fn |func )")],
        "absent": ["no code files changed", "no code constructs in scope"],
    },
    "testability": {
        "code": True,
        "paths": ["**/tests/**", "**/*.test.*", "**/conftest*"],
        "content": [("non-determinism/seam markers",
                     r"(?i)(time\.time|datetime\.now|random\.|monkeypatch|"
                     r"\bmock|singleton|\bglobal )")],
        "keywords": ["testab", "coverage", "determinis", "mockab"],
        "absent": ["no code files changed", "no test files", "no seam or "
                   "non-determinism markers"],
    },
    "design": {
        "content": [("UI markup",
                     r"(?m)(<[A-Za-z][^>\n]*>|className=|class=\"|"
                     r"<template|styled\.)")],
        "keywords": ["ux", "usability", "visual", "layout", "empty state", "loading state"],
        "absent": ["no UI component files", "no UI markup",
                   "no UX keywords in the requirement"],
    },
    "scalability": {
        "content": [
            ("SQL query surface",
             r"(?i)(select\s+.+\s+from|insert\s+into|\bjoin\b|group\s+by)"),
            ("HTTP/queue clients",
             r"(?i)(requests\.(get|post|put|delete)|urllib\.request|"
             r"\bfetch\(|axios|http\.client|aiohttp|kafka|rabbitmq|\bsqs\b|"
             r"pub/?sub|celery)"),
            ("loops over remote calls",
             r"(?is)\b(for|while)\b[^\n]*\n[^\n]{0,200}?"
             r"(select\s|\.execute\(|requests\.|fetch\(|\.query\()"),
        ],
        "graph": ["hub"],
        "keywords": ["scale", "scalab", "throughput", "latency", "hot path",
                     "load"],
        "absent": ["no api/db/services paths", "no query or client code",
                   "no remote calls in loops", "no hub module in the graph",
                   "no scalability keywords in the requirement"],
    },
    "integrability": {
        "content": [("contract/schema markers",
                     r"(?i)(openapi|swagger|protobuf|proto3|json[- ]?schema|"
                     r"content-type|api[_-]?version|/v[0-9]+/)")],
        "graph": ["boundary"],
        "keywords": ["contract", "api version", "integrat", "error code"],
        "absent": ["no api/schema/contract paths", "no contract or schema "
                   "markers", "no boundary contracts in impact"],
    },
    "data-safety": {
        "content": [("migration/DDL markers",
                     r"(?i)(alter\s+table|drop\s+(table|column)|"
                     r"add\s+column|backfill|\bmigration|not\s+null|"
                     r"on\s+delete\s+cascade)")],
        "keywords": ["migration", "rollback", "backfill", "cascade"],
        "absent": ["no migration/schema files", "no DDL or backfill markers",
                   "no migration keywords in the requirement"],
    },
    "tech-writer": {
        "content": [("doc structure",
                     r"(?m)^#{1,3}\s|^\.\. |^=====")],
        "keywords": ["readme", "changelog", "documentation", "adr"],
        "absent": ["no docs/markdown files", "no document structure",
                   "no documentation keywords in the requirement"],
    },
    "qa": {
        # D5. The qa spec carried NO path globs of its own, so the only
        # surfaces it recognized were the catalog's (`**/tests/**`,
        # `**/*.test.*`, `**/*.spec.*`, e2e/cypress/playwright/__tests__) —
        # none of which a Go repo's `pkg/cache/cache_test.go` matches. A
        # field review of a diff MADE of Go test files therefore reported
        # "no test files": an n/a that ASSERTS there was nothing to check,
        # which is the coverage-honesty feature inverted. These are the
        # by-convention test surfaces of the languages the catalog's own
        # code_extensions already admit.
        "paths": ["**/*_test.go",                      # Go
                  "**/test_*.py", "**/*_test.py",      # Python
                  "**/*.test.*", "**/*.spec.*",        # JS/TS
                  "**/*Test.java",                     # Java
                  "**/*Tests.cs",                      # C#
                  "**/*_spec.rb",                      # Ruby
                  "**/tests/**", "**/test/**", "**/spec/**"],
        # D5, second half: the construct regex was lowercase-only, so
        # Ginkgo/Gomega (`Describe(`, `It(`, `Expect(`), xUnit's `Assert`
        # and every other capitalized dialect read as prose. `i` here and
        # nowhere else — the flags are per-pattern.
        "content": [("test constructs",
                     r"(?im)(\bassert\b|expect\(|\bit\(|describe\(|"
                     r"@pytest|unittest)")],
        "keywords": ["regression", "edge case", "e2e", "test strategy"],
        "absent": ["no test files", "no test constructs",
                   "no QA keywords in the requirement"],
    },
    "devops": {
        "content": [("pipeline/build markers",
                     r"(?im)^(FROM |RUN |jobs:|steps:|stages:|pipeline\b|"
                     r"\s+uses:\s)")],
        "keywords": ["pipeline", "ci/cd", "deploy", "reproducib", "iac"],
        "absent": ["no CI/container/IaC files", "no pipeline or build "
                   "markers", "no devops keywords in the requirement"],
    },
    "dba": {
        "content": [
            ("DDL/index markers",
             r"(?i)(create\s+(table|index|unique\s+index)|alter\s+table|"
             r"foreign\s+key|primary\s+key|partition\s+by)"),
            ("query patterns",
             r"(?i)(select\s+.+\s+from|\bjoin\s|group\s+by|order\s+by)"),
            ("ORM/model markers",
             r"(?i)(models\.Model|@Entity|prisma|ActiveRecord|sqlalchemy|"
             r"@Table)"),
        ],
        "keywords": ["index", "query plan", "schema", "normaliz",
                     "partition"],
        "absent": ["no sql/models/schema files", "no DDL or index markers",
                   "no query patterns", "no ORM models"],
    },
    "sre": {
        "content": [("observability/resilience markers",
                     r"(?i)(retry|timeout|circuit[ _-]?breaker|backoff|"
                     r"prometheus|\balert|\bslo\b|healthcheck|"
                     r"health[_ ]check|runbook|pagerduty)")],
        "keywords": ["observab", "alert", "incident", "reliab", "slo",
                     "on-call"],
        "absent": ["no monitoring/alerts/runbook files", "no observability "
                   "or resilience markers", "no SRE keywords in the "
                   "requirement"],
    },
    "project-management": {
        "content": [("plan/rollout structure",
                     r"(?im)^(##\s*(milestone|timeline|rollout|risk|wave)|"
                     r"- \[ \])")],
        "keywords": ["timeline", "milestone", "cross-team", "rollout plan"],
        "absent": ["no plan/roadmap files", "no milestone or rollout "
                   "structure", "no delivery keywords in the requirement"],
    },
    "frontend": {
        "content": [
            ("component markup",
             r"(<[A-Z][A-Za-z0-9]*[\s/>]|className=|useState|useEffect|"
             r"v-if=|@Component)"),
            ("state management",
             r"(?i)(redux|zustand|vuex|pinia|useReducer|createStore)"),
            ("render/bundle perf",
             r"(?i)(React\.lazy|import\(|\bmemo\(|debounce|"
             r"requestAnimationFrame)"),
        ],
        "keywords": ["frontend", "component", "browser", "bundle"],
        "absent": ["no frontend files", "no component markup",
                   "no state management", "no render/bundle-perf markers"],
    },
    "backend": {
        "content": [
            ("route/handler markers",
             r"(?i)(@app\.(get|post|put|delete)|@router\.|app\.(get|post)\(|"
             r"HandleFunc|express\(\)|def\s+handle_)"),
            ("transaction/idempotency markers",
             r"(?i)(transaction|idempoten|\brollback\b|commit\(\)|"
             r"exactly[- ]once)"),
            ("concurrency primitives",
             r"(?i)(threading\.|asyncio|multiprocessing|semaphore|mutex|"
             r"\block\(\)|async\s+def|goroutine|sync\.WaitGroup)"),
        ],
        "graph": ["boundary"],
        "keywords": ["endpoint", "service boundar", "idempoten",
                     "business logic", "backend"],
        "absent": ["no api/services/handlers paths", "no route handlers",
                   "no transaction or idempotency markers",
                   "no concurrency primitives"],
    },
    "tradeoffs": {
        "content": [("alternatives/decision markers",
                     r"(?i)(trade[- ]?off|alternative|option [ab]\b|"
                     r"\bpros\b|\bcons\b|revisit (if|when)|we chose)")],
        "keywords": ["tradeoff", "trade-off", "alternative", "hidden cost"],
        "absent": ["no adr/design/plan files", "no alternatives or decision "
                   "markers", "no trade-off keywords in the requirement"],
    },
    "solution-design": {
        "content": [("design-contract markers",
                     r"(?i)(design contract|module boundar|"
                     r"proposed (module|graph|edge)|contract ownership|"
                     r"component diagram)")],
        "keywords": ["solution design", "design contract", "module boundar"],
        "absent": ["no design/ files", "no design-contract markers",
                   "no solution-design keywords in the requirement"],
    },
    "services-selection": {
        "content": [("dependency-manifest markers",
                     r"(?i)(\"dependencies\"|install_requires|"
                     r"\[dependencies\]|\brequire\s+['\"]|implementation\s|"
                     r"new (service|vendor|dependency))")],
        "keywords": ["vendor", "lock-in", "self-host", "managed service",
                     "new dependency"],
        "absent": ["no dependency manifests", "no dependency additions",
                   "no selection keywords in the requirement"],
    },
    "time-to-market": {
        "content": [("scope/phasing markers",
                     r"(?i)(\bmvp\b|phase [0-9]|defer(red)?\b|"
                     r"critical path|cut scope|later release)")],
        "keywords": ["deadline", "mvp", "time to market", "defer", "launch"],
        "absent": ["no plan/spec files", "no scope or phasing markers",
                   "no time-to-market keywords in the requirement"],
    },
    "architecture": {
        "content": [
            ("infra topology",
             r"(?im)^(services:|apiVersion:|resource\s+\"|module\s+\")"),
            ("architecture docs",
             r"(?i)(\badr\b|architecture|\bc4\b|component diagram|"
             r"data flow|coupling)"),
            ("service-boundary code",
             r"(?i)(grpc|proto3|message\s+\w+\s*\{|event bus|pub/?sub|"
             r"\bqueue\b)"),
        ],
        "graph": ["hub", "boundary"],
        "keywords": ["architect", "decompos", "coupling", "consistency",
                     "boundar"],
        "absent": ["no architecture/adr/infra files", "no infra topology",
                   "no architecture docs", "no service-boundary code",
                   "no hub module or boundary contract in the graph"],
    },
    "mobile": {
        "content": [
            ("platform APIs",
             r"(?i)(UIKit|SwiftUI|UIViewController|UIApplication|"
             r"android\.(os|app|content)|\bActivity\b|\bFragment\b|"
             r"\bIntent\b|CoreData|WorkManager)"),
            ("lifecycle/permissions",
             r"(?i)(onCreate|onResume|viewDidLoad|requestPermissions|"
             r"uses-permission|NSLocationWhenInUse|Info\.plist)"),
            ("offline/battery",
             r"(?i)(offline|sync adapter|battery|\bdoze\b|reachability)"),
        ],
        "keywords": ["ios", "android", "mobile", "offline", "app store"],
        "absent": ["no ios/android files", "no platform APIs",
                   "no lifecycle or permission markers",
                   "no offline/battery markers"],
    },
    "accessibility": {
        "content": [("a11y markers",
                     r"(?i)(aria-[a-z]+|role=|alt=|tabindex|screen reader|"
                     r"wcag|focus management|contrast)")],
        "keywords": ["accessib", "wcag", "aria", "keyboard nav"],
        "absent": ["no UI files", "no ARIA/alt/focus markers",
                   "no accessibility keywords in the requirement"],
    },
    "privacy-compliance": {
        "content": [("PII/consent markers",
                     r"(?i)(\bpii\b|gdpr|ccpa|consent|personal data|"
                     r"data retention|anonymi[sz]|email[_ ]address|"
                     r"\btracking\b|\banalytics\b)")],
        "keywords": ["privacy", "pii", "gdpr", "consent", "retention"],
        "absent": ["no privacy/analytics/consent paths", "no PII or consent "
                   "markers", "no privacy keywords in the requirement"],
    },
    "cost-finops": {
        "content": [("provisioning/cost markers",
                     r"(?im)(instance_type|autoscal|reserved|\begress\b|"
                     r"provisioned|^\s*(cpu|memory):\s|replicas:)")],
        "keywords": ["cost", "spend", "finops", "over-provision", "budget"],
        "absent": ["no IaC/k8s files", "no provisioning or cost markers",
                   "no cost keywords in the requirement"],
    },
    "i18n": {
        "content": [
            ("i18n imports",
             r"(?i)(import\s+[^\n]*i18n|require\(['\"](i18n|i18next)|"
             r"react-intl|formatjs|\bgettext\b|ngettext|from\s+['\"]i18n)"),
            ("locale data",
             r"(?i)(\"locale\"|\blang=|LC_ALL|setlocale|\bmsgid\b|"
             r"pluraliz|\brtl\b)"),
        ],
        "density": ("user-facing string literals", 5),
        "keywords": ["i18n", "locale", "translat", "localiz", "rtl"],
        "absent": ["no locale files", "no i18n imports",
                   "no user-facing string literals in scope"],
    },
}


# ------------------------------------------------------------ detector build

def _compiled(spec: dict):
    """Compile a spec's content rules once (cached on the spec dict)."""
    key = "_compiled"
    if key not in spec:
        spec[key] = [(label, re.compile(pat))
                     for label, pat in spec.get("content", ())]
    return spec[key]


# D-0006. A content regex is a proxy for "this code DOES x". Run it over
# prose and it becomes a proxy for "this document MENTIONS x", which is a
# different claim and usually a false one. Editing five of this repo's own
# documentation files fired seventeen lenses — `dba` went DEEP because
# routing-and-flows.md explains query patterns, and `data-safety` fired on
# the privacy LENS DEFINITION, a file whose entire job is to describe
# migration markers so a reviewer can spot them.
#
# The fix is not "never score markdown": tech-writer, product and
# solution-design have documentation as their real surface. It is that a
# content marker in a prose file only counts for a lens whose OWN declared
# surface admits that file. tech-writer's globs say `**/*.md`, so it keeps
# scoring; dba's say nothing of the sort, so it stops. Path and requirement
# signals are untouched — a doc that a lens's globs claim still routes it.
PROSE_EXT = (".md", ".mdx", ".markdown", ".rst", ".txt", ".adoc")


def _is_prose(rel: str) -> bool:
    return str(rel or "").lower().endswith(PROSE_EXT)


def _density_hits(ctx: Ctx, code_ext) -> tuple[int, int, int, str]:
    """(count, files, real_count, real_rel) of user-facing-looking string
    literals (>= 3 words) across changed code files; real_count counts only
    files the D-0002 discount does NOT apply to (fixture-class and not
    B5-exempt), and real_rel names the first such file."""
    total, nfiles, real, real_rel = 0, 0, 0, ""
    for rel, text in ctx.contents():
        if not _is_code(rel, code_ext):
            continue
        n = len(_STR_LIT.findall(text))
        if n:
            total += n
            nfiles += 1
            if not ctx.is_discounted(rel):
                real += n
                real_rel = real_rel or rel
    return total, nfiles, real, real_rel


def _spec_detect(lens_id: str, spec: dict, catalog_lens: dict,
                 cat: dict, ctx: Ctx) -> dict:
    evidence = []
    score = 0.0

    # -- path signal: catalog globs + spec extras (+ code extensions when
    #    the lens's surface is "any code change")
    globs = sorted(set((catalog_lens.get("globs") or [])
                       + list(spec.get("paths", ()))))
    # `is_discounted` is `is_fixture_path` minus the B5 product-dir
    # exemption, so full-weight support is a SUPERSET of what it was.
    real_files = [f for f in ctx.files if not ctx.is_discounted(f)]
    hit = _glob_hit(ctx.files, globs) if globs else None
    if hit:
        real_hit = _glob_hit(real_files, globs)
        if real_hit:
            evidence.append(f"path: {hit[0]} matches {hit[1]}"
                            + ctx.exemption_note(real_hit[0]))
            score += W_PATH
        else:   # ONLY fixture-class support -> re-weight, never suppress
            evidence.append(f"path: {hit[0]} matches {hit[1]} "
                            f"{_DISCOUNT_NOTE}")
            score += W_PATH * FIXTURE_DISCOUNT
    elif spec.get("code"):
        code_ext = cat.get("code_extensions") or []
        code_files = [f for f in ctx.files if _is_code(f, code_ext)]
        if code_files:
            label = (f"path: code change ({code_files[0]}"
                     + (f" +{len(code_files) - 1} more" if
                        len(code_files) > 1 else "") + ")")
            real_code = [f for f in code_files if not ctx.is_discounted(f)]
            if real_code:
                evidence.append(label + ctx.exemption_note(real_code[0]))
                score += W_PATH
            else:
                evidence.append(f"{label} {_DISCOUNT_NOTE}")
                score += W_PATH * FIXTURE_DISCOUNT

    # Lenses 2.0: absence can itself be applicability evidence. QA must see
    # a production-code change that carries no test path; adding the reason
    # only in lens.route was too late because this engine had already
    # returned n/a. Give the trigger normal path-signal weight so it routes
    # light without manufacturing a deep verdict.
    if (catalog_lens.get("untested_trigger")
            and _change_adds_no_test(ctx.files,
                                    cat.get("code_extensions") or [])):
        evidence.append("change shape: code changed with no test file")
        score += W_PATH

    # -- content signals (bounded corpus, first hit per rule)
    #
    # D-0006: prose is scanned only for a lens whose own surface claims it.
    # `globs` above is exactly that surface (catalog globs + spec extras),
    # so this needs no second list to drift out of sync.
    def _scannable(rel):
        return not _is_prose(rel) or bool(globs and _glob_hit([rel], globs))

    for label, rx in _compiled(spec):
        found = None
        real_rel = None
        for rel, text in ctx.contents():
            if not _scannable(rel):
                continue
            if rx.search(text):
                if found is None:
                    found = rel
                if not ctx.is_discounted(rel):
                    real_rel = rel
                    break
        if found:
            if real_rel is not None:
                evidence.append(f"content: {label} in {found}"
                                + ctx.exemption_note(real_rel))
                score += W_CONTENT
            else:   # ONLY fixture-class support
                evidence.append(f"content: {label} in {found} "
                                f"{_DISCOUNT_NOTE}")
                score += W_CONTENT * FIXTURE_DISCOUNT

    # -- density signal
    dens = spec.get("density")
    if dens:
        label, threshold = dens
        count, nfiles, real_count, real_rel = _density_hits(
            ctx, cat.get("code_extensions") or [])
        if count >= threshold:
            if real_count:
                evidence.append(f"content: {label}: {count} across "
                                f"{nfiles} file(s)"
                                + ctx.exemption_note(real_rel))
                score += W_DENSITY
            else:   # ONLY fixture-class support
                evidence.append(f"content: {label}: {count} across "
                                f"{nfiles} file(s) {_DISCOUNT_NOTE}")
                score += W_DENSITY * FIXTURE_DISCOUNT

    # -- requirement-text keywords
    kws = sorted(k for k in spec.get("keywords", ())
                 if k in ctx.requirement_text)
    if kws:
        evidence.append("requirement: mentions " + ", ".join(kws))
        score += W_KEYWORD

    # -- graph flags
    for flag in spec.get("graph", ()):
        if flag == "hub":
            hub = int(ctx.graph.get("hub_dependents") or 0)
            if hub >= _HUB_DEPENDENTS:
                evidence.append(f"graph: hub module ({hub} direct "
                                "dependents)")
                score += W_GRAPH
        elif flag == "boundary":
            bcs = sorted(ctx.graph.get("boundary_contracts") or [])
            if bcs:
                evidence.append("graph: boundary contracts in impact: "
                                + ", ".join(bcs[:3]))
                score += W_GRAPH

    score = round(min(1.0, score), 4)
    negative = []
    if score < LIGHT:
        if evidence:
            negative = [f"0 {lens_id} signals strong enough (score {score} "
                        f"< {LIGHT}): only {len(evidence)} weak indicator(s); "
                        + ", ".join(spec["absent"][:2])]
        else:
            negative = [f"0 {lens_id} signals: "
                        + ", ".join(spec["absent"])]
    return {"score": score, "evidence": evidence,
            "negative_evidence": negative}


def _make_detector(lens_id: str, spec: dict, catalog_lens: dict, cat: dict):
    def detector(ctx: Ctx) -> dict:
        return _spec_detect(lens_id, spec, catalog_lens, cat, ctx)
    detector.__name__ = f"detect_{lens_id.replace('-', '_')}"
    return detector


def _build_registry() -> dict:
    cat = load_catalog()
    by_id = {l["id"]: l for l in cat["lenses"]}
    missing = sorted(set(by_id) - set(SPECS))
    extra = sorted(set(SPECS) - set(by_id))
    if missing or extra:
        # fail closed at import: a catalog/spec drift must never silently
        # route a lens with no detector (or a detector with no lens)
        raise ValueError(f"lens_signals spec drift: missing={missing} "
                         f"extra={extra}")
    return {lid: _make_detector(lid, SPECS[lid], by_id[lid], cat)
            for lid in sorted(by_id)}


DETECTORS: dict = _build_registry()


def requirement_keyword_lenses(ctx: Ctx, lens_ids=None) -> dict:
    """The requirement-keyword detector, re-runnable on its own.

    {lens_id: [matched keywords]} for every lens whose spec keywords appear
    in ctx.requirement_text — the SAME `k in ctx.requirement_text` rule
    `_spec_detect` scores with W_KEYWORD, isolated so a caller holding only
    a ctx can ask "which lenses does this requirement's own text earn?".

    B4 (R-0008): a component's cached lens_map is derived WITHOUT
    requirement_text, so lens.py's component assembly re-runs this LIVE on
    the ctx it already builds and UNIONS the result into the proposed set —
    a cached map may add candidates, never subtract them. Empty
    requirement text -> {} (no widening, byte-unchanged routing)."""
    if not ctx.requirement_text:
        return {}
    ids = sorted(SPECS) if lens_ids is None else \
        sorted(set(lens_ids) & set(SPECS))
    out: dict = {}
    for lid in ids:
        kws = sorted(k for k in SPECS[lid].get("keywords", ())
                     if k in ctx.requirement_text)
        if kws:
            out[lid] = kws
    return out


# ----------------------------------------------------------------- verdicts

def detect(lens_id: str, ctx: Ctx) -> dict:
    """Run one registered detector and validate its result shape.
    Unknown lens or malformed result -> ValueError (fail closed)."""
    det = DETECTORS.get(lens_id)
    if det is None:
        raise ValueError(f"unknown lens id: {lens_id!r} (catalog has "
                         f"{len(DETECTORS)} registered detectors)")
    r = det(ctx)
    if not isinstance(r, dict):
        raise ValueError(f"detector {lens_id}: result must be a dict")
    try:
        score = float(r["score"])
    except (KeyError, TypeError, ValueError):
        raise ValueError(f"detector {lens_id}: missing/non-numeric score")
    if not (0.0 <= score <= 1.0):
        raise ValueError(f"detector {lens_id}: score {score} outside 0..1")
    ev = r.get("evidence")
    neg = r.get("negative_evidence")
    if not isinstance(ev, list) or not isinstance(neg, list) \
            or not all(isinstance(x, str) for x in ev + neg):
        raise ValueError(f"detector {lens_id}: evidence/negative_evidence "
                         "must be lists of strings")
    return {"score": score, "evidence": list(ev), "negative_evidence":
            list(neg)}


def verdict_for_score(score: float) -> str:
    if score >= DEEP:
        return "deep"
    if score >= LIGHT:
        return "light"
    return "n/a"


def verdicts(lens_ids, ctx: Ctx) -> dict:
    """Describe lens relevance using only observed source signals."""
    out = {}
    for lid in sorted(set(lens_ids)):
        result = detect(lid, ctx)
        out[lid] = {"verdict": verdict_for_score(result["score"]),
                    "score": result["score"], "evidence": result["evidence"],
                    "negative_evidence": result["negative_evidence"]}
    return out


def route_verdicts(workspace, files, stage=None, requirement_text=None,
                   graph=None, content_by_file=None) -> dict:
    """Suggest relevant lenses without dispatching workers or imposing quotas."""
    ctx = make_ctx(workspace, files, requirement_text=requirement_text,
                   graph=graph, stage=stage, content_by_file=content_by_file)
    return verdicts([lens["id"] for lens in load_catalog()["lenses"]], ctx)

SHA-256: 25a0d6b935bcb48434175d1a38b5995e4cc3f9216f041afc1ec43575771aecbe