← Files taskplaneARCHIVED FILE
taskplane/graph_primitives.py
52.2 KB · Oct 2, 2026 · 00:29 UTC
"""Lower-owned graph identities, context records, and lens applicability.
This module is the single dependency-free contract shared by the graph
scanner, component decomposition, and lens facade. It owns the deterministic
component lens engine so direct graph APIs need no higher-layer activation.
"""
from __future__ import annotations
import json
import os
import posixpath
import re
if __package__:
from . import glob_match, path_roles
else:
import glob_match
import path_roles
_is_test_path = path_roles.is_test_path
_change_adds_no_test = path_roles.change_adds_no_test
# Directory names that mark source layout rather than component identity.
_SRC_ROOTS = ("src", "app", "lib", "packages", "pkg", "internal", "cmd")
_GO_MODULE_LINE = re.compile(r"^\s*module\s+(\S+)", re.M)
_ID_PREFIXES = ("ext:", "svc:", "req:", "contract:", "resource:")
# Reserved manifest-map key for the repository's own Go module path. The
# leading NUL cannot collide with a real repository directory.
ROOT_MODULE_KEY = "\x00root_module"
# Edges whose direction means "from NEEDS to". Structural and annotation
# edges deliberately do not answer dependent/blast-radius questions.
DEPENDENCY_EDGE_KINDS = frozenset({
"imports", "depends_on", "consumes", "depends", "calls", "uses",
})
_FIXTURE_SEGMENTS = frozenset({"fixtures", "testdata", "goldens"})
_GRAPH_LOADER = None
def register_graph_loader(loader) -> None:
"""Register the scanner-owned read boundary used by lens projection."""
global _GRAPH_LOADER
_GRAPH_LOADER = loader
def load_graph(workspace: str) -> dict:
"""Read through the registered persistence boundary.
Persistence remains owned by depgraph; this lower layer only holds the
injected callable, which keeps lens routing from importing the scanner.
"""
if _GRAPH_LOADER is None:
raise RuntimeError("graph loader is not registered")
return _GRAPH_LOADER(workspace)
def is_fixture_module(module: str) -> bool:
"""Whether a module boundary itself is fixture-classed."""
value = str(module or "").replace("\\", "/").strip("/")
if not value or value == "(root)":
return False
return (any(part.lower() in _FIXTURE_SEGMENTS
for part in value.split("/") if part)
or value.lower().endswith(".golden"))
def manifest_modules(files, read) -> dict:
"""Return ``{directory: declared import identity}`` for owned manifests.
Only package.json names and go.mod module paths are import identities.
Root package manifests describe the repository, while a root go.mod path
is retained solely as an import prefix under ``ROOT_MODULE_KEY``.
"""
out: dict = {}
for rel in sorted(files or ()):
rel = str(rel).replace("\\", "/")
base = posixpath.basename(rel)
if base not in ("package.json", "go.mod"):
continue
directory = posixpath.dirname(rel)
if not directory:
if base == "go.mod":
match = _GO_MODULE_LINE.search(read(rel) or "")
if match and match.group(1) and "/" in match.group(1):
out[ROOT_MODULE_KEY] = match.group(1).strip()
continue
text = read(rel)
if not text:
continue
declared = None
if base == "package.json":
try:
data = json.loads(text)
except (ValueError, TypeError):
continue
if isinstance(data, dict) and isinstance(data.get("name"), str):
declared = data["name"].strip()
else:
match = _GO_MODULE_LINE.search(text)
declared = match.group(1).strip() if match else None
if not declared or declared.startswith(_ID_PREFIXES):
continue
out[directory] = declared.replace("\\", "/").strip("/")
return {key: value for key, value in out.items() if value}
def declared_module_ids(graph: dict | None) -> dict:
"""Return the manifest identity map persisted by the last graph scan."""
return ((graph or {}).get("meta") or {}).get("module_ids") or {}
def root_module(declared_ids) -> "str | None":
"""Return the repository Go module prefix when held in map form."""
if isinstance(declared_ids, dict):
return declared_ids.get(ROOT_MODULE_KEY) or None
return None
def strip_root_prefix(spec: str, root: "str | None") -> "str | None":
"""Map ``<root>/pkg/x`` to the repository-relative ``pkg/x``."""
if not root:
return None
spec = str(spec or "").replace("\\", "/").strip("/")
if spec == root:
return None
if spec.startswith(root + "/"):
return spec[len(root) + 1:] or None
return None
def _strip_root_module(spec: str, declared_ids) -> "str | None":
return strip_root_prefix(spec, root_module(declared_ids))
def _declared_target(spec: str, declared_ids) -> "str | None":
"""Return the longest declared module matching an import specifier."""
if not declared_ids:
return None
spec = str(spec or "").replace("\\", "/").strip("/")
while spec:
if spec in declared_ids and spec != ROOT_MODULE_KEY:
return spec
if "/" not in spec:
return None
spec = spec.rsplit("/", 1)[0]
return None
def module_of(relpath: str, manifests: dict | None = None) -> str:
"""Return the stable module identity owning a repository-relative path."""
relpath = str(relpath or "").replace("\\", "/")
directory = posixpath.dirname(relpath)
if manifests:
probe = directory
while probe:
hit = manifests.get(probe)
if hit:
return hit
parent = posixpath.dirname(probe)
if parent == probe:
break
probe = parent
if not directory:
return "(root)"
parts = directory.split("/")
for marker in (("src", "main", "java"), ("src", "test", "java")):
for index in range(0, len(parts) - len(marker) + 1):
if tuple(parts[index:index + len(marker)]) == marker:
package = parts[index + len(marker):]
if package:
return "/".join(package[-3:])
kept = [part for part in parts if part not in _SRC_ROOTS]
return "/".join(kept[:2]) if kept else parts[-1]
def node_kind(node: str) -> str:
"""Return the public graph-node family for an identifier."""
if node.startswith("component:"):
return "component"
if node.startswith("surface:"):
return "surface"
if node.startswith("req:"):
return "requirement"
if node.startswith("contract:"):
return "contract"
if node.startswith("resource:"):
return "resource"
if node.startswith("svc:"):
return "infra"
if node.startswith("ext:"):
return "external"
return "module"
def is_boundary(node: str) -> bool:
return node.startswith(("contract:", "resource:", "svc:", "ext:"))
def is_dependency_edge(edge: dict) -> bool:
"""Whether ``edge`` expresses the dependency direction from NEEDS to."""
try:
return edge.get("kind") in DEPENDENCY_EDGE_KINDS
except AttributeError:
return False
def strongly_connected_components(nodes, edges) -> list[list[str]]:
"""Return a deterministic, complete Tarjan decomposition.
``edges`` contains ``(source, target)`` pairs. Unknown endpoints are not
silently admitted: the caller must validate them before asking for SCC
evidence. This helper never truncates; bounded callers must refuse input
over their declared ceiling before calling it.
"""
ordered = sorted({str(node) for node in nodes or ()})
known = set(ordered)
adjacency = {node: set() for node in ordered}
for source, target in edges or ():
source, target = str(source), str(target)
if source not in known or target not in known:
raise ValueError(
f"SCC edge names an unknown node: {source} -> {target}")
adjacency[source].add(target)
index = 0
indexes: dict[str, int] = {}
lowlinks: dict[str, int] = {}
stack: list[str] = []
on_stack: set[str] = set()
components: list[list[str]] = []
def visit(node: str) -> None:
nonlocal index
indexes[node] = lowlinks[node] = index
index += 1
stack.append(node)
on_stack.add(node)
for target in sorted(adjacency[node]):
if target not in indexes:
visit(target)
lowlinks[node] = min(lowlinks[node], lowlinks[target])
elif target in on_stack:
lowlinks[node] = min(lowlinks[node], indexes[target])
if lowlinks[node] != indexes[node]:
return
component = []
while stack:
member = stack.pop()
on_stack.remove(member)
component.append(member)
if member == node:
break
components.append(sorted(component))
for node in ordered:
if node not in indexes:
visit(node)
return sorted(components, key=lambda component: tuple(component))
def graph_payload(graph: dict, modules,
*, fixture_module_predicate=None) -> dict:
"""Project raw graph data into the shared lens/component context record.
``hub_dependents`` intentionally counts every incoming edge for backward
compatibility. ``module_dependents`` is stricter: only dependency edges,
excluding self-dependence and fixture-class witnesses. The result stays
an ordinary dict so all existing JSON and caller payloads remain stable.
"""
selected = sorted({str(module) for module in modules or () if module})
selected_set = set(selected)
incoming: dict[str, set] = {}
dependency_incoming: dict[str, set] = {}
contracts: set[str] = set()
fixture_module_predicate = (
fixture_module_predicate or (lambda _value: False))
for edge in (graph or {}).get("edges") or []:
if not isinstance(edge, dict):
continue
source, target = edge.get("from"), edge.get("to")
if not source or not target:
continue
incoming.setdefault(target, set()).add(source)
if is_dependency_edge(edge):
dependency_incoming.setdefault(target, set()).add(source)
for left, right in ((source, target), (target, source)):
if left in selected_set and str(right).startswith("contract:"):
contracts.add(str(right))
return {
"hub_dependents": max(
(len(incoming.get(module, ())) for module in selected), default=0),
"boundary_contracts": sorted(contracts),
"modules": selected,
"module_ids": declared_module_ids(graph),
"module_dependents": {
module: len([
dependent
for dependent in dependency_incoming.get(module, ())
if dependent != module
and not fixture_module_predicate(str(dependent))
])
for module in selected
},
}
# ---------------------------------------------------------------- thresholds
DEEP = 0.6 # score >= DEEP -> "deep"
LIGHT = 0.2 # score >= LIGHT -> "light"; below -> "n/a"
MAX_FILE_BYTES = 64 * 1024 # per-file content-scan bound
MAX_FILES = 200 # max files content-scanned per ctx
# signal weights (sum is clamped to 1.0)
W_PATH = 0.35 # a changed path matches the lens's surface globs
W_CONTENT = 0.3 # a content marker fires in a changed file
W_DENSITY = 0.25 # density-style content signal (e.g. user-facing strings)
W_KEYWORD = 0.15 # the requirement text mentions the lens's concern
W_GRAPH = 0.35 # graph flag (hub module / boundary contract in impact)
_HUB_DEPENDENTS = 3 # >= this many direct dependents -> hub signal fires
# ------------------------------------------- fixtures-path discount (D-0002)
#
# Checked-in test fixtures LOOK like the surfaces they imitate (a locale
# file under tests/fixtures/ scores like a real locale file), so fixture-only
# diffs inflated i18n/mobile to deep (D-0002). Path/content/density signal
# hits whose ONLY support is fixture-class files are RE-WEIGHTED x0.25 —
# never suppressed: the evidence line survives and names the discount
# (honesty: the evidence says why the score is low), n/a semantics are
# untouched, and the floors still apply after scoring. A hit with any real
# product-file support keeps full weight.
FIXTURE_DISCOUNT = 0.25
_FIXTURE_SEGMENTS = frozenset({"fixtures", "testdata", "goldens"})
_FIXTURE_EXTENSIONS = (".golden",)
_DISCOUNT_NOTE = f"(fixture-path discount x{FIXTURE_DISCOUNT})"
def is_fixture_path(path: str) -> bool:
"""True when the path is fixture-class: any path segment named
fixtures/testdata/goldens (at any depth), or a .golden extension."""
p = str(path).replace(os.sep, "/")
if any(seg.lower() in _FIXTURE_SEGMENTS for seg in p.split("/") if seg):
return True
return p.lower().endswith(_FIXTURE_EXTENSIONS)
# B5 (R-0008): the classifier above keys off the PATH alone, so a REAL
# product directory literally named `fixtures/` was discounted — the
# dangerous direction (under-routing). The discount APPLICATION point now
# consults a graph-informed exemption computed at ctx construction: a
# fixture-classed path whose containing graph MODULE has at least
# FIXTURE_EXEMPT_MIN_DEPENDENTS dependents is real product code (nothing
# depends on a test-fixture tree), keeps FULL weight, and says so in the
# evidence line. The exemption can only ever RESTORE weight — the
# discounted set strictly shrinks, never grows — and is_fixture_path()
# itself is untouched, so every other caller sees the same classification.
#
# TWO guards keep that premise TRUE, because the first cut of this exemption
# repealed D-0002 on this very repository:
#
# 1. WHAT COUNTS AS A DEPENDENT — only depgraph.DEPENDENCY_EDGE_KINDS
# ("from NEEDS to"). `module_dependents` used to count EVERY incoming
# edge, including the STRUCTURAL `defined_in` edge a docker-compose file
# emits ABOUT ITS OWN module. This repo's
# taskplane/tests/fixtures/detectors/architecture/positive/
# docker-compose.yml therefore handed module `taskplane/tests` two
# "dependents" that were the fixtures themselves. Filtered in
# _graph_payload here AND in decompose._graph_payload, so cached
# component maps and live routing agree.
#
# 2. AT WHAT GRANULARITY — a graph module id is at most a few segments
# (`taskplane/tests`) while is_fixture_path classifies the FULL path
# (`taskplane/tests/fixtures/detectors/i18n/positive/locales/en.json`).
# An ancestor module id must never speak for a fixture subtree nested
# below it, so the exemption fires only when the fixture classification
# is visible AT the module boundary that carries the dependent count —
# is_fixture_path(module) is itself True (`fixtures`, `tests/fixtures`,
# …). When the fixture-marking segment lies BELOW the module id, the
# dependents belong to the module, not to the fixture tree, and the
# discount stands. On this repo all 102 fixture-classed tracked paths
# map to non-fixture-classed modules (taskplane/tests, src, api,
# components, auth), so the whole corpus stays discounted.
#
# Both guards only SHRINK the exemption, and the exemption only ever restores
# weight, so behaviour stays between "always discounted" (base) and the
# unguarded version: it can never route anything more narrowly than base.
FIXTURE_EXEMPT_MIN_DEPENDENTS = 1
def _module_is_fixture_classed(module: str) -> bool:
"""Guard 2: the fixture marking must be visible at the module boundary
that carries the dependent count, not somewhere deeper in the path."""
return is_fixture_module(module)
def _module_of(path: str, manifests: dict | None = None) -> str:
"""The graph module id owning ``path`` through the shared contract."""
p = str(path).replace(os.sep, "/")
try:
return module_of(p, manifests)
except Exception:
d = os.path.dirname(p)
return "/".join(d.split("/")[:2]) if d else "(root)"
def _fixture_exemptions(files, graph) -> dict:
"""{path: reason} — the fixture-classed paths that keep FULL weight.
Guard 2 (granularity) is applied here; guard 1 (dependency-edge kinds)
is applied where `module_dependents` is BUILT (_graph_payload)."""
deps = (graph or {}).get("module_dependents")
if not isinstance(deps, dict) or not deps:
return {}
ids = (graph or {}).get("module_ids") or None
out: dict = {}
for rel in files:
if not is_fixture_path(rel):
continue
module = _module_of(rel, ids)
if not _module_is_fixture_classed(module):
# the fixture segment sits BELOW the module boundary: this
# module's dependents do not describe the fixture tree
continue
try:
n = int(deps.get(module, 0) or 0)
except (TypeError, ValueError):
n = 0
if n >= FIXTURE_EXEMPT_MIN_DEPENDENTS:
out[rel] = (f"product-dir exemption: {rel} is in module "
f"{module}, which has {n} dependent(s) — real "
"product code, not a test fixture")
return out
# ------------------------------------------------------------------- catalog
_CATALOG_CACHE: dict = {}
def _plugin_root() -> str:
# lenses/ sits at the plugin root, one level up from taskplane/
return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
def load_catalog(root: str | None = None) -> dict:
"""Load lenses/catalog.json (cached per root). Self-contained on purpose:
lens.py will import THIS module in t2, so importing lens here would set
up an import cycle."""
key = root or _plugin_root()
if key not in _CATALOG_CACHE:
with open(os.path.join(key, "lenses", "catalog.json"), encoding="utf-8") as f:
_CATALOG_CACHE[key] = json.load(f)
return _CATALOG_CACHE[key]
# -------------------------------------------------------------- glob matcher
def _match(path: str, glob: str) -> bool:
"""Compatibility facade over the shared dependency-neutral matcher."""
return glob_match.path_matches(path, glob)
def _glob_hit(files, globs):
"""First (file, glob) pair that matches, or None. files/globs iterated
in the given (already sorted) order -> deterministic."""
return glob_match.first_match(files, globs)
def _is_code(path: str, code_ext) -> bool:
return any(path.endswith(e) for e in code_ext)
# ------------------------------------------------------------------- context
class Ctx:
"""Everything a detector may look at. Bounded, cached, deterministic.
files changed paths relative to workspace (sorted, deduped)
workspace absolute-ish root used to read file contents
requirement_text lowercased requirement/acceptance-criteria blob
graph {"hub_dependents": int, "boundary_contracts": [str],
"modules": [str], "module_dependents": {mod: int}}
stage delivery stage (context only; unused by detectors)
fixture_exempt {path: reason} — fixture-classed paths that keep FULL
weight because their graph module has dependents (B5,
R-0008); computed HERE, at ctx construction
"""
__slots__ = ("workspace", "files", "requirement_text", "graph", "stage",
"fixture_exempt", "_contents", "_content_by_file")
def __init__(self, workspace, files, requirement_text, graph, stage,
content_by_file=None):
self.workspace = workspace
self.files = sorted({str(f).replace(os.sep, "/") for f in files or []})
if isinstance(requirement_text, (list, tuple)):
requirement_text = "\n".join(str(x) for x in requirement_text)
self.requirement_text = (requirement_text or "").lower()
self.graph = graph or {"hub_dependents": 0,
"boundary_contracts": [], "modules": []}
self.stage = stage
self.fixture_exempt = _fixture_exemptions(self.files, self.graph)
self._content_by_file = (
{str(path).replace(os.sep, "/"): str(text)
for path, text in content_by_file.items()
if str(path).replace(os.sep, "/") in self.files}
if isinstance(content_by_file, dict) else None)
self._contents = None
def is_discounted(self, path: str) -> bool:
"""True when the D-0002 fixture discount APPLIES to `path`: it is
fixture-class AND not graph-exempt (B5). A strict subset of
is_fixture_path() — the exemption can only restore weight."""
p = str(path).replace(os.sep, "/")
return is_fixture_path(p) and p not in self.fixture_exempt
def exemption_note(self, path: str) -> str:
"""' (reason)' when `path` is exempt from the discount, else ''."""
reason = self.fixture_exempt.get(str(path).replace(os.sep, "/"))
return f" ({reason})" if reason else ""
def read(self, relpath: str) -> str | None:
"""Bounded read of one changed file: at most MAX_FILE_BYTES bytes,
decoded utf-8 with replacement. Missing/unreadable -> None (changed
lists legitimately contain deletions)."""
p = os.path.join(self.workspace, relpath)
try:
# Containment (EM v3): changed-file lists come from git, but a
# crafted relpath ('../..') or a symlink pointing outside the
# workspace must not let a detector read foreign files. Resolve
# and require the real target stays under the real workspace.
root = os.path.realpath(self.workspace)
real = os.path.realpath(p)
if real != root and not real.startswith(root + os.sep):
return None
with open(real, "rb") as f:
text = f.read(MAX_FILE_BYTES).decode("utf-8", "replace")
# Read as bytes (deliberately — no locale codec, no newline
# translation), then normalize line endings ourselves. Detector
# regexes are line-anchored and the scores they produce are
# frozen in goldens; a file checked out with CRLF must score
# identically to the same file with LF, or the same diff routes
# differently on Windows than it does in CI.
return text.replace("\r\n", "\n").replace("\r", "\n")
except OSError:
return None
def contents(self):
"""[(relpath, text)] for the first MAX_FILES sorted changed files
that exist. Cached: the corpus is read once per ctx, then every
detector scans the same in-memory snapshot."""
if self._contents is None:
out = []
for rel in self.files:
if len(out) >= MAX_FILES:
break
if self._content_by_file is not None:
text = self._content_by_file.get(rel)
if text is not None:
text = text.encode("utf-8")[:MAX_FILE_BYTES].decode(
"utf-8", "replace")
else:
text = self.read(rel)
if text is not None:
out.append((rel, text))
self._contents = out
return self._contents
def make_ctx(workspace, files, requirement_text=None, graph=None,
stage=None, content_by_file=None) -> Ctx:
"""Build source-signal context; unavailable graphs yield an empty payload."""
if graph is None:
graph = _graph_payload(workspace, files)
return Ctx(workspace, files, requirement_text, graph, stage,
content_by_file=content_by_file)
def _graph_payload(workspace, files) -> dict:
try:
g = load_graph(workspace)
# the graph's own declared ids, or the fixture-exemption and
# dependent-count lookups below miss every workspace module
_ids = declared_module_ids(g)
touched = sorted({module_of(
str(f).replace(os.sep, "/"), _ids)
for f in files or []})
return graph_payload(
g, touched,
fixture_module_predicate=is_fixture_module)
except Exception:
return {"hub_dependents": 0, "boundary_contracts": [], "modules": [],
"module_dependents": {}}
# --------------------------------------------------------- signal-spec table
#
# One spec per catalog lens. Shape:
# paths: extra surface globs ON TOP of the lens's catalog globs
# (the catalog globs are always part of the path signal)
# code: True -> the path signal also fires on any code-extension file
# (baseline lenses: their surface IS "code changed")
# content: [(label, regex_pattern)] scanned over the bounded corpus;
# each distinct rule that fires adds W_CONTENT
# density: (label, threshold) -> user-facing string-literal density rule
# graph: subset of {"hub", "boundary"}
# keywords: substrings looked up in the lowercased requirement text
# absent: negative-evidence phrases, joined into
# "0 <lens> signals: no X, no Y, ..." when nothing fires
_STR_LIT = re.compile(r"""["']([A-Za-z][A-Za-z,.!?'-]*(?:\s+[A-Za-z][A-Za-z,.!?'-]*){2,})["']""")
SPECS: dict[str, dict] = {
"product": {
"content": [("spec/acceptance markers",
r"(?im)^#+.*(acceptance criteria|user stor|requirement)")],
"keywords": ["user journey", "acceptance", "success metric",
"user value"],
"absent": ["no spec/requirements files", "no acceptance-criteria "
"markers", "no product keywords in the requirement"],
},
"security": {
"paths": ["**/hooks/**", "**/taskplane_lite.py", "**/*login*",
"**/*permission*", "**/*.pem", "**/*credential*"],
"content": [
("auth/secret markers",
r"(?i)(password|passwd|secret[_a-z]*\s*=|api[_-]?key|"
r"bearer\s|jwt|oauth|csrf|bcrypt|hmac|authenticat|authoriz|"
r"permission|session[_ ]token)"),
("unsafe-input surface",
r"(?i)(\beval\(|\bexec\(|subprocess|os\.system|pickle\.loads|"
r"yaml\.load\(|innerHTML|dangerouslySetInnerHTML|"
r"shell\s*=\s*True)"),
],
"graph": ["boundary"],
"keywords": ["auth", "security", "secret", "permission", "vulnerab",
"enforc", "injection"],
"absent": ["no auth/secrets/enforcement paths", "no auth or secret "
"markers", "no unsafe-input surface", "no boundary "
"contracts in impact", "no security keywords in the "
"requirement"],
},
"code-quality": {
"code": True,
"content": [("code constructs",
r"(?m)^\s*(def |class |function\b|const |public |"
r"private |fn |func )")],
"absent": ["no code files changed", "no code constructs in scope"],
},
"testability": {
"code": True,
"paths": ["**/tests/**", "**/*.test.*", "**/conftest*"],
"content": [("non-determinism/seam markers",
r"(?i)(time\.time|datetime\.now|random\.|monkeypatch|"
r"\bmock|singleton|\bglobal )")],
"keywords": ["testab", "coverage", "determinis", "mockab"],
"absent": ["no code files changed", "no test files", "no seam or "
"non-determinism markers"],
},
"design": {
"content": [("UI markup",
r"(?m)(<[A-Za-z][^>\n]*>|className=|class=\"|"
r"<template|styled\.)")],
"keywords": ["ux", "usability", "visual", "layout", "empty state", "loading state"],
"absent": ["no UI component files", "no UI markup",
"no UX keywords in the requirement"],
},
"scalability": {
"content": [
("SQL query surface",
r"(?i)(select\s+.+\s+from|insert\s+into|\bjoin\b|group\s+by)"),
("HTTP/queue clients",
r"(?i)(requests\.(get|post|put|delete)|urllib\.request|"
r"\bfetch\(|axios|http\.client|aiohttp|kafka|rabbitmq|\bsqs\b|"
r"pub/?sub|celery)"),
("loops over remote calls",
r"(?is)\b(for|while)\b[^\n]*\n[^\n]{0,200}?"
r"(select\s|\.execute\(|requests\.|fetch\(|\.query\()"),
],
"graph": ["hub"],
"keywords": ["scale", "scalab", "throughput", "latency", "hot path",
"load"],
"absent": ["no api/db/services paths", "no query or client code",
"no remote calls in loops", "no hub module in the graph",
"no scalability keywords in the requirement"],
},
"integrability": {
"content": [("contract/schema markers",
r"(?i)(openapi|swagger|protobuf|proto3|json[- ]?schema|"
r"content-type|api[_-]?version|/v[0-9]+/)")],
"graph": ["boundary"],
"keywords": ["contract", "api version", "integrat", "error code"],
"absent": ["no api/schema/contract paths", "no contract or schema "
"markers", "no boundary contracts in impact"],
},
"data-safety": {
"content": [("migration/DDL markers",
r"(?i)(alter\s+table|drop\s+(table|column)|"
r"add\s+column|backfill|\bmigration|not\s+null|"
r"on\s+delete\s+cascade)")],
"keywords": ["migration", "rollback", "backfill", "cascade"],
"absent": ["no migration/schema files", "no DDL or backfill markers",
"no migration keywords in the requirement"],
},
"tech-writer": {
"content": [("doc structure",
r"(?m)^#{1,3}\s|^\.\. |^=====")],
"keywords": ["readme", "changelog", "documentation", "adr"],
"absent": ["no docs/markdown files", "no document structure",
"no documentation keywords in the requirement"],
},
"qa": {
# D5. The qa spec carried NO path globs of its own, so the only
# surfaces it recognized were the catalog's (`**/tests/**`,
# `**/*.test.*`, `**/*.spec.*`, e2e/cypress/playwright/__tests__) —
# none of which a Go repo's `pkg/cache/cache_test.go` matches. A
# field review of a diff MADE of Go test files therefore reported
# "no test files": an n/a that ASSERTS there was nothing to check,
# which is the coverage-honesty feature inverted. These are the
# by-convention test surfaces of the languages the catalog's own
# code_extensions already admit.
"paths": ["**/*_test.go", # Go
"**/test_*.py", "**/*_test.py", # Python
"**/*.test.*", "**/*.spec.*", # JS/TS
"**/*Test.java", # Java
"**/*Tests.cs", # C#
"**/*_spec.rb", # Ruby
"**/tests/**", "**/test/**", "**/spec/**"],
# D5, second half: the construct regex was lowercase-only, so
# Ginkgo/Gomega (`Describe(`, `It(`, `Expect(`), xUnit's `Assert`
# and every other capitalized dialect read as prose. `i` here and
# nowhere else — the flags are per-pattern.
"content": [("test constructs",
r"(?im)(\bassert\b|expect\(|\bit\(|describe\(|"
r"@pytest|unittest)")],
"keywords": ["regression", "edge case", "e2e", "test strategy"],
"absent": ["no test files", "no test constructs",
"no QA keywords in the requirement"],
},
"devops": {
"content": [("pipeline/build markers",
r"(?im)^(FROM |RUN |jobs:|steps:|stages:|pipeline\b|"
r"\s+uses:\s)")],
"keywords": ["pipeline", "ci/cd", "deploy", "reproducib", "iac"],
"absent": ["no CI/container/IaC files", "no pipeline or build "
"markers", "no devops keywords in the requirement"],
},
"dba": {
"content": [
("DDL/index markers",
r"(?i)(create\s+(table|index|unique\s+index)|alter\s+table|"
r"foreign\s+key|primary\s+key|partition\s+by)"),
("query patterns",
r"(?i)(select\s+.+\s+from|\bjoin\s|group\s+by|order\s+by)"),
("ORM/model markers",
r"(?i)(models\.Model|@Entity|prisma|ActiveRecord|sqlalchemy|"
r"@Table)"),
],
"keywords": ["index", "query plan", "schema", "normaliz",
"partition"],
"absent": ["no sql/models/schema files", "no DDL or index markers",
"no query patterns", "no ORM models"],
},
"sre": {
"content": [("observability/resilience markers",
r"(?i)(retry|timeout|circuit[ _-]?breaker|backoff|"
r"prometheus|\balert|\bslo\b|healthcheck|"
r"health[_ ]check|runbook|pagerduty)")],
"keywords": ["observab", "alert", "incident", "reliab", "slo",
"on-call"],
"absent": ["no monitoring/alerts/runbook files", "no observability "
"or resilience markers", "no SRE keywords in the "
"requirement"],
},
"project-management": {
"content": [("plan/rollout structure",
r"(?im)^(##\s*(milestone|timeline|rollout|risk|wave)|"
r"- \[ \])")],
"keywords": ["timeline", "milestone", "cross-team", "rollout plan"],
"absent": ["no plan/roadmap files", "no milestone or rollout "
"structure", "no delivery keywords in the requirement"],
},
"frontend": {
"content": [
("component markup",
r"(<[A-Z][A-Za-z0-9]*[\s/>]|className=|useState|useEffect|"
r"v-if=|@Component)"),
("state management",
r"(?i)(redux|zustand|vuex|pinia|useReducer|createStore)"),
("render/bundle perf",
r"(?i)(React\.lazy|import\(|\bmemo\(|debounce|"
r"requestAnimationFrame)"),
],
"keywords": ["frontend", "component", "browser", "bundle"],
"absent": ["no frontend files", "no component markup",
"no state management", "no render/bundle-perf markers"],
},
"backend": {
"content": [
("route/handler markers",
r"(?i)(@app\.(get|post|put|delete)|@router\.|app\.(get|post)\(|"
r"HandleFunc|express\(\)|def\s+handle_)"),
("transaction/idempotency markers",
r"(?i)(transaction|idempoten|\brollback\b|commit\(\)|"
r"exactly[- ]once)"),
("concurrency primitives",
r"(?i)(threading\.|asyncio|multiprocessing|semaphore|mutex|"
r"\block\(\)|async\s+def|goroutine|sync\.WaitGroup)"),
],
"graph": ["boundary"],
"keywords": ["endpoint", "service boundar", "idempoten",
"business logic", "backend"],
"absent": ["no api/services/handlers paths", "no route handlers",
"no transaction or idempotency markers",
"no concurrency primitives"],
},
"tradeoffs": {
"content": [("alternatives/decision markers",
r"(?i)(trade[- ]?off|alternative|option [ab]\b|"
r"\bpros\b|\bcons\b|revisit (if|when)|we chose)")],
"keywords": ["tradeoff", "trade-off", "alternative", "hidden cost"],
"absent": ["no adr/design/plan files", "no alternatives or decision "
"markers", "no trade-off keywords in the requirement"],
},
"solution-design": {
"content": [("design-contract markers",
r"(?i)(design contract|module boundar|"
r"proposed (module|graph|edge)|contract ownership|"
r"component diagram)")],
"keywords": ["solution design", "design contract", "module boundar"],
"absent": ["no design/ files", "no design-contract markers",
"no solution-design keywords in the requirement"],
},
"services-selection": {
"content": [("dependency-manifest markers",
r"(?i)(\"dependencies\"|install_requires|"
r"\[dependencies\]|\brequire\s+['\"]|implementation\s|"
r"new (service|vendor|dependency))")],
"keywords": ["vendor", "lock-in", "self-host", "managed service",
"new dependency"],
"absent": ["no dependency manifests", "no dependency additions",
"no selection keywords in the requirement"],
},
"time-to-market": {
"content": [("scope/phasing markers",
r"(?i)(\bmvp\b|phase [0-9]|defer(red)?\b|"
r"critical path|cut scope|later release)")],
"keywords": ["deadline", "mvp", "time to market", "defer", "launch"],
"absent": ["no plan/spec files", "no scope or phasing markers",
"no time-to-market keywords in the requirement"],
},
"architecture": {
"content": [
("infra topology",
r"(?im)^(services:|apiVersion:|resource\s+\"|module\s+\")"),
("architecture docs",
r"(?i)(\badr\b|architecture|\bc4\b|component diagram|"
r"data flow|coupling)"),
("service-boundary code",
r"(?i)(grpc|proto3|message\s+\w+\s*\{|event bus|pub/?sub|"
r"\bqueue\b)"),
],
"graph": ["hub", "boundary"],
"keywords": ["architect", "decompos", "coupling", "consistency",
"boundar"],
"absent": ["no architecture/adr/infra files", "no infra topology",
"no architecture docs", "no service-boundary code",
"no hub module or boundary contract in the graph"],
},
"mobile": {
"content": [
("platform APIs",
r"(?i)(UIKit|SwiftUI|UIViewController|UIApplication|"
r"android\.(os|app|content)|\bActivity\b|\bFragment\b|"
r"\bIntent\b|CoreData|WorkManager)"),
("lifecycle/permissions",
r"(?i)(onCreate|onResume|viewDidLoad|requestPermissions|"
r"uses-permission|NSLocationWhenInUse|Info\.plist)"),
("offline/battery",
r"(?i)(offline|sync adapter|battery|\bdoze\b|reachability)"),
],
"keywords": ["ios", "android", "mobile", "offline", "app store"],
"absent": ["no ios/android files", "no platform APIs",
"no lifecycle or permission markers",
"no offline/battery markers"],
},
"accessibility": {
"content": [("a11y markers",
r"(?i)(aria-[a-z]+|role=|alt=|tabindex|screen reader|"
r"wcag|focus management|contrast)")],
"keywords": ["accessib", "wcag", "aria", "keyboard nav"],
"absent": ["no UI files", "no ARIA/alt/focus markers",
"no accessibility keywords in the requirement"],
},
"privacy-compliance": {
"content": [("PII/consent markers",
r"(?i)(\bpii\b|gdpr|ccpa|consent|personal data|"
r"data retention|anonymi[sz]|email[_ ]address|"
r"\btracking\b|\banalytics\b)")],
"keywords": ["privacy", "pii", "gdpr", "consent", "retention"],
"absent": ["no privacy/analytics/consent paths", "no PII or consent "
"markers", "no privacy keywords in the requirement"],
},
"cost-finops": {
"content": [("provisioning/cost markers",
r"(?im)(instance_type|autoscal|reserved|\begress\b|"
r"provisioned|^\s*(cpu|memory):\s|replicas:)")],
"keywords": ["cost", "spend", "finops", "over-provision", "budget"],
"absent": ["no IaC/k8s files", "no provisioning or cost markers",
"no cost keywords in the requirement"],
},
"i18n": {
"content": [
("i18n imports",
r"(?i)(import\s+[^\n]*i18n|require\(['\"](i18n|i18next)|"
r"react-intl|formatjs|\bgettext\b|ngettext|from\s+['\"]i18n)"),
("locale data",
r"(?i)(\"locale\"|\blang=|LC_ALL|setlocale|\bmsgid\b|"
r"pluraliz|\brtl\b)"),
],
"density": ("user-facing string literals", 5),
"keywords": ["i18n", "locale", "translat", "localiz", "rtl"],
"absent": ["no locale files", "no i18n imports",
"no user-facing string literals in scope"],
},
}
# ------------------------------------------------------------ detector build
def _compiled(spec: dict):
"""Compile a spec's content rules once (cached on the spec dict)."""
key = "_compiled"
if key not in spec:
spec[key] = [(label, re.compile(pat))
for label, pat in spec.get("content", ())]
return spec[key]
# D-0006. A content regex is a proxy for "this code DOES x". Run it over
# prose and it becomes a proxy for "this document MENTIONS x", which is a
# different claim and usually a false one. Editing five of this repo's own
# documentation files fired seventeen lenses — `dba` went DEEP because
# routing-and-flows.md explains query patterns, and `data-safety` fired on
# the privacy LENS DEFINITION, a file whose entire job is to describe
# migration markers so a reviewer can spot them.
#
# The fix is not "never score markdown": tech-writer, product and
# solution-design have documentation as their real surface. It is that a
# content marker in a prose file only counts for a lens whose OWN declared
# surface admits that file. tech-writer's globs say `**/*.md`, so it keeps
# scoring; dba's say nothing of the sort, so it stops. Path and requirement
# signals are untouched — a doc that a lens's globs claim still routes it.
PROSE_EXT = (".md", ".mdx", ".markdown", ".rst", ".txt", ".adoc")
def _is_prose(rel: str) -> bool:
return str(rel or "").lower().endswith(PROSE_EXT)
def _density_hits(ctx: Ctx, code_ext) -> tuple[int, int, int, str]:
"""(count, files, real_count, real_rel) of user-facing-looking string
literals (>= 3 words) across changed code files; real_count counts only
files the D-0002 discount does NOT apply to (fixture-class and not
B5-exempt), and real_rel names the first such file."""
total, nfiles, real, real_rel = 0, 0, 0, ""
for rel, text in ctx.contents():
if not _is_code(rel, code_ext):
continue
n = len(_STR_LIT.findall(text))
if n:
total += n
nfiles += 1
if not ctx.is_discounted(rel):
real += n
real_rel = real_rel or rel
return total, nfiles, real, real_rel
def _spec_detect(lens_id: str, spec: dict, catalog_lens: dict,
cat: dict, ctx: Ctx) -> dict:
evidence = []
score = 0.0
# -- path signal: catalog globs + spec extras (+ code extensions when
# the lens's surface is "any code change")
globs = sorted(set((catalog_lens.get("globs") or [])
+ list(spec.get("paths", ()))))
# `is_discounted` is `is_fixture_path` minus the B5 product-dir
# exemption, so full-weight support is a SUPERSET of what it was.
real_files = [f for f in ctx.files if not ctx.is_discounted(f)]
hit = _glob_hit(ctx.files, globs) if globs else None
if hit:
real_hit = _glob_hit(real_files, globs)
if real_hit:
evidence.append(f"path: {hit[0]} matches {hit[1]}"
+ ctx.exemption_note(real_hit[0]))
score += W_PATH
else: # ONLY fixture-class support -> re-weight, never suppress
evidence.append(f"path: {hit[0]} matches {hit[1]} "
f"{_DISCOUNT_NOTE}")
score += W_PATH * FIXTURE_DISCOUNT
elif spec.get("code"):
code_ext = cat.get("code_extensions") or []
code_files = [f for f in ctx.files if _is_code(f, code_ext)]
if code_files:
label = (f"path: code change ({code_files[0]}"
+ (f" +{len(code_files) - 1} more" if
len(code_files) > 1 else "") + ")")
real_code = [f for f in code_files if not ctx.is_discounted(f)]
if real_code:
evidence.append(label + ctx.exemption_note(real_code[0]))
score += W_PATH
else:
evidence.append(f"{label} {_DISCOUNT_NOTE}")
score += W_PATH * FIXTURE_DISCOUNT
# Lenses 2.0: absence can itself be applicability evidence. QA must see
# a production-code change that carries no test path; adding the reason
# only in lens.route was too late because this engine had already
# returned n/a. Give the trigger normal path-signal weight so it routes
# light without manufacturing a deep verdict.
if (catalog_lens.get("untested_trigger")
and _change_adds_no_test(ctx.files,
cat.get("code_extensions") or [])):
evidence.append("change shape: code changed with no test file")
score += W_PATH
# -- content signals (bounded corpus, first hit per rule)
#
# D-0006: prose is scanned only for a lens whose own surface claims it.
# `globs` above is exactly that surface (catalog globs + spec extras),
# so this needs no second list to drift out of sync.
def _scannable(rel):
return not _is_prose(rel) or bool(globs and _glob_hit([rel], globs))
for label, rx in _compiled(spec):
found = None
real_rel = None
for rel, text in ctx.contents():
if not _scannable(rel):
continue
if rx.search(text):
if found is None:
found = rel
if not ctx.is_discounted(rel):
real_rel = rel
break
if found:
if real_rel is not None:
evidence.append(f"content: {label} in {found}"
+ ctx.exemption_note(real_rel))
score += W_CONTENT
else: # ONLY fixture-class support
evidence.append(f"content: {label} in {found} "
f"{_DISCOUNT_NOTE}")
score += W_CONTENT * FIXTURE_DISCOUNT
# -- density signal
dens = spec.get("density")
if dens:
label, threshold = dens
count, nfiles, real_count, real_rel = _density_hits(
ctx, cat.get("code_extensions") or [])
if count >= threshold:
if real_count:
evidence.append(f"content: {label}: {count} across "
f"{nfiles} file(s)"
+ ctx.exemption_note(real_rel))
score += W_DENSITY
else: # ONLY fixture-class support
evidence.append(f"content: {label}: {count} across "
f"{nfiles} file(s) {_DISCOUNT_NOTE}")
score += W_DENSITY * FIXTURE_DISCOUNT
# -- requirement-text keywords
kws = sorted(k for k in spec.get("keywords", ())
if k in ctx.requirement_text)
if kws:
evidence.append("requirement: mentions " + ", ".join(kws))
score += W_KEYWORD
# -- graph flags
for flag in spec.get("graph", ()):
if flag == "hub":
hub = int(ctx.graph.get("hub_dependents") or 0)
if hub >= _HUB_DEPENDENTS:
evidence.append(f"graph: hub module ({hub} direct "
"dependents)")
score += W_GRAPH
elif flag == "boundary":
bcs = sorted(ctx.graph.get("boundary_contracts") or [])
if bcs:
evidence.append("graph: boundary contracts in impact: "
+ ", ".join(bcs[:3]))
score += W_GRAPH
score = round(min(1.0, score), 4)
negative = []
if score < LIGHT:
if evidence:
negative = [f"0 {lens_id} signals strong enough (score {score} "
f"< {LIGHT}): only {len(evidence)} weak indicator(s); "
+ ", ".join(spec["absent"][:2])]
else:
negative = [f"0 {lens_id} signals: "
+ ", ".join(spec["absent"])]
return {"score": score, "evidence": evidence,
"negative_evidence": negative}
def _make_detector(lens_id: str, spec: dict, catalog_lens: dict, cat: dict):
def detector(ctx: Ctx) -> dict:
return _spec_detect(lens_id, spec, catalog_lens, cat, ctx)
detector.__name__ = f"detect_{lens_id.replace('-', '_')}"
return detector
def _build_registry() -> dict:
cat = load_catalog()
by_id = {l["id"]: l for l in cat["lenses"]}
missing = sorted(set(by_id) - set(SPECS))
extra = sorted(set(SPECS) - set(by_id))
if missing or extra:
# fail closed at import: a catalog/spec drift must never silently
# route a lens with no detector (or a detector with no lens)
raise ValueError(f"lens_signals spec drift: missing={missing} "
f"extra={extra}")
return {lid: _make_detector(lid, SPECS[lid], by_id[lid], cat)
for lid in sorted(by_id)}
DETECTORS: dict = _build_registry()
def requirement_keyword_lenses(ctx: Ctx, lens_ids=None) -> dict:
"""The requirement-keyword detector, re-runnable on its own.
{lens_id: [matched keywords]} for every lens whose spec keywords appear
in ctx.requirement_text — the SAME `k in ctx.requirement_text` rule
`_spec_detect` scores with W_KEYWORD, isolated so a caller holding only
a ctx can ask "which lenses does this requirement's own text earn?".
B4 (R-0008): a component's cached lens_map is derived WITHOUT
requirement_text, so lens.py's component assembly re-runs this LIVE on
the ctx it already builds and UNIONS the result into the proposed set —
a cached map may add candidates, never subtract them. Empty
requirement text -> {} (no widening, byte-unchanged routing)."""
if not ctx.requirement_text:
return {}
ids = sorted(SPECS) if lens_ids is None else \
sorted(set(lens_ids) & set(SPECS))
out: dict = {}
for lid in ids:
kws = sorted(k for k in SPECS[lid].get("keywords", ())
if k in ctx.requirement_text)
if kws:
out[lid] = kws
return out
# ----------------------------------------------------------------- verdicts
def detect(lens_id: str, ctx: Ctx) -> dict:
"""Run one registered detector and validate its result shape.
Unknown lens or malformed result -> ValueError (fail closed)."""
det = DETECTORS.get(lens_id)
if det is None:
raise ValueError(f"unknown lens id: {lens_id!r} (catalog has "
f"{len(DETECTORS)} registered detectors)")
r = det(ctx)
if not isinstance(r, dict):
raise ValueError(f"detector {lens_id}: result must be a dict")
try:
score = float(r["score"])
except (KeyError, TypeError, ValueError):
raise ValueError(f"detector {lens_id}: missing/non-numeric score")
if not (0.0 <= score <= 1.0):
raise ValueError(f"detector {lens_id}: score {score} outside 0..1")
ev = r.get("evidence")
neg = r.get("negative_evidence")
if not isinstance(ev, list) or not isinstance(neg, list) \
or not all(isinstance(x, str) for x in ev + neg):
raise ValueError(f"detector {lens_id}: evidence/negative_evidence "
"must be lists of strings")
return {"score": score, "evidence": list(ev), "negative_evidence":
list(neg)}
def verdict_for_score(score: float) -> str:
if score >= DEEP:
return "deep"
if score >= LIGHT:
return "light"
return "n/a"
def verdicts(lens_ids, ctx: Ctx) -> dict:
"""Describe lens relevance using only observed source signals."""
out = {}
for lid in sorted(set(lens_ids)):
result = detect(lid, ctx)
out[lid] = {"verdict": verdict_for_score(result["score"]),
"score": result["score"], "evidence": result["evidence"],
"negative_evidence": result["negative_evidence"]}
return out
def route_verdicts(workspace, files, stage=None, requirement_text=None,
graph=None, content_by_file=None) -> dict:
"""Suggest relevant lenses without dispatching workers or imposing quotas."""
ctx = make_ctx(workspace, files, requirement_text=requirement_text,
graph=graph, stage=stage, content_by_file=content_by_file)
return verdicts([lens["id"] for lens in load_catalog()["lenses"]], ctx)
SHA-256: 25a0d6b935bcb48434175d1a38b5995e4cc3f9216f041afc1ec43575771aecbe