← Files Repo ScoutARCHIVED FILE

skills/repo-scout/scripts/inventory.py

17.8 KB · Oct 5, 2026 · 18:33 UTC

↓ Download file

#!/usr/bin/env python3
# SPDX-License-Identifier: GPL-3.0-or-later
# Copyright (C) 2026 ImL1s
"""Bounded, filename-only repository inventory. Never executes repository code.

Python 3.10+, standard library only. Output is JSON on stdout; no file writes,
network calls, dependency installation, or subprocesses. This is not an analyzer.
Exit 0: inventory traversal finished within policy; 2: partial traversal.
"""
from __future__ import annotations

import argparse
from collections import Counter, defaultdict
from datetime import datetime, timezone
import json
import os
from pathlib import Path
import stat
import sys
from typing import Any

TOOL_VERSION = "0.1.2"  # kept equal to skills/repo-scout/VERSION by tests/test_package.py

SENSITIVE_DIRS = {".ssh", ".aws", ".secrets", ".gnupg"}
CONVENIENCE_DIRS = {
    ".git", ".hg", ".svn", "node_modules", ".venv", "venv", "__pycache__",
    ".gradle", ".idea", ".vs", ".next", ".nuxt", ".svelte-kit", ".turbo",
    ".dart_tool", ".terraform", "DerivedData", "Pods", "build", "dist",
    "target", "coverage", "vendor",
}
EXCLUDED_DIRS = SENSITIVE_DIRS | CONVENIENCE_DIRS
LANGUAGES = {
    ".py": "Python", ".pyi": "Python", ".js": "JavaScript", ".mjs": "JavaScript",
    ".cjs": "JavaScript", ".jsx": "JavaScript/JSX", ".ts": "TypeScript",
    ".tsx": "TypeScript/TSX", ".vue": "Vue", ".svelte": "Svelte",
    ".html": "HTML", ".css": "CSS", ".scss": "SCSS", ".less": "Less",
    ".dart": "Dart", ".kt": "Kotlin", ".kts": "Kotlin script", ".java": "Java",
    ".swift": "Swift", ".m": "Objective-C or MATLAB (ambiguous)", ".mm": "Objective-C++",
    ".go": "Go", ".rs": "Rust", ".c": "C", ".h": "C/C++ header (ambiguous)",
    ".cc": "C++", ".cpp": "C++", ".cxx": "C++", ".hpp": "C++",
    ".cs": "C#", ".fs": "F#", ".vb": "Visual Basic", ".php": "PHP",
    ".rb": "Ruby", ".ex": "Elixir", ".exs": "Elixir", ".erl": "Erlang",
    ".hrl": "Erlang", ".scala": "Scala", ".clj": "Clojure", ".cljs": "ClojureScript",
    ".hs": "Haskell", ".ml": "OCaml", ".mli": "OCaml", ".lua": "Lua",
    ".zig": "Zig", ".jl": "Julia", ".r": "R", ".sql": "SQL", ".sh": "Shell",
    ".bash": "Shell", ".zsh": "Shell", ".ps1": "PowerShell", ".bat": "Windows batch",
    ".sol": "Solidity", ".move": "Move", ".tf": "Terraform", ".proto": "Protocol Buffers",
    ".graphql": "GraphQL", ".gql": "GraphQL", ".elm": "Elm",
    ".s": "Assembly", ".asm": "Assembly", ".v": "Verilog or Coq (ambiguous)",
    ".vhd": "VHDL", ".vhdl": "VHDL", ".f90": "Fortran", ".f95": "Fortran",
    ".cob": "COBOL", ".cbl": "COBOL", ".rkt": "Racket", ".pl": "Perl or Prolog (ambiguous)",
    ".ipynb": "Jupyter notebook", ".yaml": "YAML", ".yml": "YAML", ".json": "JSON",
    ".toml": "TOML", ".xml": "XML", ".md": "Markdown", ".mdx": "MDX",
}
MARKERS = {
    "package.json": "javascript-package", "pnpm-workspace.yaml": "javascript-workspace",
    "pnpm-lock.yaml": "javascript-lockfile", "yarn.lock": "javascript-lockfile",
    "package-lock.json": "javascript-lockfile", "bun.lock": "javascript-lockfile",
    "pubspec.yaml": "dart-package", "pubspec.lock": "dart-lockfile",
    "build.gradle": "gradle-project", "build.gradle.kts": "gradle-project",
    "settings.gradle": "gradle-workspace", "settings.gradle.kts": "gradle-workspace",
    "AndroidManifest.xml": "android", "Package.swift": "swift-package",
    "project.pbxproj": "xcode-project", "Podfile": "cocoapods-project",
    "Cargo.toml": "rust-package", "Cargo.lock": "rust-lockfile", "go.mod": "go-module",
    "go.work": "go-workspace", "pyproject.toml": "python-project", "setup.py": "python-project",
    "requirements.txt": "python-dependencies", "uv.lock": "python-lockfile",
    "poetry.lock": "python-lockfile", "pom.xml": "maven-project",
    "Gemfile": "ruby-project", "composer.json": "php-project", "mix.exs": "elixir-project",
    "build.sbt": "scala-project", "dune-project": "ocaml-project",
    "cabal.project": "haskell-project", "Project.toml": "julia-project",
    "build.zig": "zig-project", "CMakeLists.txt": "cmake-project",
    "meson.build": "meson-project", "Makefile": "make-project",
    "platformio.ini": "embedded-project", "Dockerfile": "container-build",
    "compose.yaml": "container-compose", "docker-compose.yml": "container-compose",
    "Chart.yaml": "helm-chart", "foundry.toml": "foundry-project",
    "AGENTS.md": "agent-instructions", "CLAUDE.md": "agent-instructions",
}
MARKER_SUFFIXES = {
    ".csproj": "dotnet-project", ".fsproj": "dotnet-project", ".vbproj": "dotnet-project",
    ".sln": "dotnet-workspace", ".cabal": "haskell-package", ".gemspec": "ruby-package",
    ".tf": "terraform-file", ".proto": "protobuf-contract", ".graphql": "graphql-contract",
}


def sensitive_name(name: str) -> bool:
    lower = name.lower()
    return (lower == ".env" or lower.startswith(".env.")
            or lower in {"id_rsa", "id_ed25519", "credentials", "credentials.json"}
            or lower.endswith((".pem", ".p12", ".pfx", ".key", ".keystore", ".jks")))


def scope_path(raw: str) -> tuple[str, ...]:
    """Split one repo-relative, /-separated scope path into literal components.

    Nothing is trimmed: " private " and "private" are different directories, so a component
    with leading or trailing whitespace is rejected rather than silently retargeted. A
    trailing separator ("a/b/") is still accepted as naming the directory."""
    text = raw.replace("\\", "/") if sys.platform == "win32" else raw
    if not text:
        raise ValueError("Scope paths must not be empty")
    if text.startswith("/") or (len(text) > 1 and text[1] == ":"):
        raise ValueError(f"Scope paths must be repo-relative, not absolute: {raw!r}")
    parts = text.split("/")
    while parts and parts[-1] == "":
        parts.pop()
    if not parts or any(part in ("", ".", "..") for part in parts):
        raise ValueError(f"Scope paths must not contain empty, '.' or '..' components: {raw!r}")
    if any(part != part.strip() for part in parts):
        raise ValueError(f"Scope path components must not start or end with whitespace: {raw!r}")
    return tuple(parts)


# Windows reparse points: a junction is not a symlink to is_symlink() but redirects a path
# exactly like one. Any reparse tag with the name-surrogate bit set is a redirect; other
# reparse points (cloud placeholders, compressed files) are the entry itself.
_FILE_ATTRIBUTE_REPARSE_POINT = getattr(stat, "FILE_ATTRIBUTE_REPARSE_POINT", 0x400)
_REPARSE_NAME_SURROGATE = 0x20000000


def is_redirect_stat(st: os.stat_result) -> bool:
    """True if no-follow metadata describes a symlink or a Windows junction-like reparse point."""
    if stat.S_ISLNK(st.st_mode):
        return True
    if sys.platform == "win32" and getattr(st, "st_file_attributes", 0) & _FILE_ATTRIBUTE_REPARSE_POINT:
        tag = getattr(st, "st_reparse_tag", 0)
        return tag == 0 or bool(tag & _REPARSE_NAME_SURROGATE)
    return False


def is_prefix(ancestor: tuple[str, ...], path: tuple[str, ...]) -> bool:
    """Component-wise ancestor-or-equal test; 'foo' never matches 'foobar'."""
    return path[:len(ancestor)] == ancestor


def parse_scope(include: list[str] | tuple[str, ...] = (),
                exclude: list[str] | tuple[str, ...] = ()) -> tuple[list[tuple[str, ...]], list[tuple[str, ...]]]:
    includes = sorted({scope_path(raw) for raw in include})
    excludes = sorted({scope_path(raw) for raw in exclude})
    for inc in includes:
        for ex in excludes:
            if is_prefix(ex, inc):
                raise ValueError(f"--include {'/'.join(inc)} is inside --exclude {'/'.join(ex)}; excludes win")
        if any(part.lower() in SENSITIVE_DIRS for part in inc):
            raise ValueError(f"--include {'/'.join(inc)} enters a hard-excluded sensitive directory")
        if sensitive_name(inc[-1]):
            raise ValueError(f"--include {'/'.join(inc)} names a sensitive file that is never inspected")
    return includes, excludes


def inventory(root: Path, max_files: int = 40000, max_entries: int = 100000,
              include: list[str] | tuple[str, ...] = (),
              exclude: list[str] | tuple[str, ...] = ()) -> dict[str, Any]:
    """Read entry metadata only; skipped directory names are visible policy, not coverage.

    Scope precedence, highest first: root boundary / symlinks and junctions never followed /
    non-regular entries / hard sensitive names > --exclude > --include > convenience exclusions.
    --include is additive: it re-admits paths under convenience-excluded directories by
    passing through the necessary ancestors only. Nothing reads .gitignore."""
    root = root.expanduser().resolve(strict=True)
    if not root.is_dir():
        raise ValueError("Repository path must be a directory")
    if max_files <= 0 or max_entries <= 0:
        raise ValueError("Limits must be positive")
    if root == Path(root.anchor):
        raise ValueError("Refusing a filesystem-root scan; provide a repository directory")
    includes, excludes = parse_scope(include, exclude)
    counts: Counter[str] = Counter()
    extensions: Counter[str] = Counter()
    skipped: Counter[str] = Counter()
    samples: dict[str, list[str]] = defaultdict(list)
    marker_counts: Counter[str] = Counter()
    markers: list[dict[str, str]] = []
    omissions: list[dict[str, str]] = []
    failures: list[dict[str, str]] = []
    include_matched: set[tuple[str, ...]] = set()
    exclude_matched: set[tuple[str, ...]] = set()
    file_count = 0
    entry_count = 0
    # (directory, components relative to root, restricted-to-include-paths)
    pending: list[tuple[Path, tuple[str, ...], bool]] = [(root, (), False)]
    entry_limit_hit = False
    file_limit_hit = False

    def omit(path: str, reason: str) -> None:
        skipped[reason] += 1
        if len(omissions) < 100:
            omissions.append({"path": path, "reason": reason})

    def excluded_by_user(parts: tuple[str, ...]) -> bool:
        hits = [ex for ex in excludes if is_prefix(ex, parts)]
        exclude_matched.update(hits)
        return bool(hits)

    def on_include_path(parts: tuple[str, ...]) -> bool:
        hits = [inc for inc in includes if is_prefix(inc, parts)]
        include_matched.update(hits)
        return bool(hits)

    def toward_include(parts: tuple[str, ...]) -> bool:
        return any(is_prefix(parts, inc) and parts != inc for inc in includes)

    while pending and not file_limit_hit and not entry_limit_hit:
        directory, dir_parts, restricted = pending.pop()
        entries = []
        try:
            with os.scandir(directory) as stream:
                for entry in stream:
                    if entry_count >= max_entries:
                        entry_limit_hit = True
                        break
                    entry_count += 1
                    entries.append(entry)
        except OSError as error:
            failures.append({"path": "/".join(dir_parts) or ".", "error": type(error).__name__})
            continue
        for entry in sorted(entries, key=lambda e: e.name):
            parts = dir_parts + (entry.name,)
            relative = "/".join(parts)
            try:
                # One no-follow stat classifies the entry; symlinks and junctions are never followed.
                st = entry.stat(follow_symlinks=False)
                if is_redirect_stat(st):
                    omit(relative, "symlink-not-followed")
                    continue
                if stat.S_ISDIR(st.st_mode):
                    if entry.name.lower() in SENSITIVE_DIRS:
                        omit(relative, "sensitive-directory-not-inspected")
                    elif excluded_by_user(parts):
                        omit(relative, "user-excluded")
                    elif on_include_path(parts):
                        pending.append((Path(entry.path), parts, False))
                    elif toward_include(parts):
                        # Pass through toward an include. Only a convenience-excluded ancestor
                        # (or one already restricted) limits what else is inventoried on the way;
                        # an ordinary ancestor keeps its normal, unrestricted traversal.
                        pending.append((Path(entry.path), parts, restricted or entry.name in CONVENIENCE_DIRS))
                    elif restricted:
                        omit(relative, "outside-include-scope")
                    elif entry.name in CONVENIENCE_DIRS:
                        omit(relative, "excluded-directory-name")
                    else:
                        pending.append((Path(entry.path), parts, False))
                    continue
                if not stat.S_ISREG(st.st_mode):
                    omit(relative, "non-regular-file")
                    continue
                if sensitive_name(entry.name):
                    omit(relative, "sensitive-filename-not-inspected")
                    continue
                if excluded_by_user(parts):
                    omit(relative, "user-excluded")
                    continue
                if not on_include_path(parts) and restricted:
                    omit(relative, "outside-include-scope")
                    continue
                if file_count >= max_files:
                    file_limit_hit = True
                    break
                file_count += 1
                suffix = Path(entry.name).suffix.lower() or "[no extension]"
                language = LANGUAGES.get(suffix, "unclassified")
                counts[language] += 1
                extensions[suffix] += 1
                if len(samples[language]) < 4:
                    samples[language].append(relative)
                marker = MARKERS.get(entry.name) or MARKER_SUFFIXES.get(suffix)
                if not marker and entry.name.upper().startswith("README"):
                    marker = "project-documentation"
                if not marker and relative.startswith(".github/workflows/") and suffix in {".yml", ".yaml"}:
                    marker = "github-workflow"
                if marker:
                    marker_counts[marker] += 1
                    if len(markers) < 1000:
                        markers.append({"path": relative, "signal": marker})
            except OSError as error:
                failures.append({"path": relative, "error": type(error).__name__})
    incomplete = []
    if entry_limit_hit:
        incomplete.append("entry_limit")
    if file_limit_hit:
        incomplete.append("file_limit")
    if failures:
        incomplete.append("filesystem_errors")
    return {
        "schema_version": "1.1", "tool_version": TOOL_VERSION,
        "generated_at": datetime.now(timezone.utc).isoformat(),
        "root": str(root), "status": "inventory_only",
        "traversal_partial": bool(incomplete), "incomplete_reasons": incomplete,
        "files_counted": file_count, "entries_enumerated": entry_count,
        "scope": {
            "include": ["/".join(inc) for inc in includes],
            "exclude": ["/".join(ex) for ex in excludes],
            "include_mode": "additive",
            "include_unmatched": ["/".join(inc) for inc in includes if inc not in include_matched],
            "exclude_unmatched": ["/".join(ex) for ex in excludes if ex not in exclude_matched],
            "hard_excluded_directory_names": sorted(SENSITIVE_DIRS),
            "convenience_excluded_directory_names": sorted(CONVENIENCE_DIRS),
        },
        "gitignore_respected": False,
        "language_hints_by_filename": dict(sorted(counts.items())),
        "extension_counts": dict(sorted(extensions.items())),
        "file_samples": dict(sorted(samples.items())),
        "marker_counts": dict(sorted(marker_counts.items())),
        "marker_samples": sorted(markers, key=lambda x: x["path"]),
        "marker_samples_truncated": sum(marker_counts.values()) > len(markers),
        "skipped_counts": dict(sorted(skipped.items())),
        "skipped_path_samples": omissions,
        "skipped_path_samples_truncated": sum(skipped.values()) > len(omissions),
        "filesystem_errors": failures[:100],
        "filesystem_error_count": len(failures),
        "limits": {"max_files": max_files, "max_entries": max_entries},
        "limitations": [
            "Filenames only: no source, manifests, imports, dependency versions, or behavior were inspected.",
            "A language or manifest hint is not parser support, framework detection, or verification.",
            "No repository commands, tests, git operations, network requests, or writes were performed.",
            "Excluded names may contain first-party code; review omissions and inspect relevant paths separately.",
            "Scope flags are additive CLI arguments; .gitignore and other ignore files are never read.",
            "Symlinks and junctions (never followed), submodule contents hidden by exclusions, and concurrent filesystem changes can limit coverage.",
            "This bounded traversal is not a security sandbox or an atomic filesystem snapshot.",
        ],
    }


def main(argv: list[str] | None = None) -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("repo", type=Path)
    parser.add_argument("--max-files", type=int, default=40000)
    parser.add_argument("--max-entries", type=int, default=100000)
    parser.add_argument("--include", action="append", default=[], metavar="PATH",
                        help="repo-relative file or subtree to admit even under a convenience-excluded directory (repeatable, additive)")
    parser.add_argument("--exclude", action="append", default=[], metavar="PATH",
                        help="repo-relative file or subtree to skip (repeatable; wins over --include; "
                             "a path that matched nothing is listed under scope.exclude_unmatched)")
    args = parser.parse_args(argv)
    try:
        report = inventory(args.repo, args.max_files, args.max_entries, args.include, args.exclude)
    except (OSError, ValueError) as error:
        parser.error(str(error))
    json.dump(report, sys.stdout, ensure_ascii=False, indent=2)
    print()
    return 2 if report["traversal_partial"] else 0


if __name__ == "__main__":
    raise SystemExit(main())

SHA-256: 0d62c5de2b8314e4454f1527eba6d01192dd71f611ce9f936d5a4ba463d42d50