← Files Biohub ESMARCHIVED FILE

scripts/biohub_esm_lib/security.py

29.7 KB · Sep 30, 2026 · 23:14 UTC

↓ Download file

"""Credential presence reporting and recursive redaction."""

from __future__ import annotations

import itertools
import os
import platform
import re
import subprocess
import threading
from collections.abc import Mapping
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, Callable

from .constants import BIOHUB_API_KEY_URL, BIOHUB_QUICKSTART_URL

MACOS_KEYCHAIN_SERVICE = "com.openai.codex.biohub-esm.ESM_API_KEY"
KEYCHAIN_LOOKUP_TIMEOUT_SECONDS = 2.0
DISABLE_KEYCHAIN_ENV = "BIOHUB_ESM_DISABLE_KEYCHAIN"
_RUNTIME_REDACTION_SECRETS: set[str] = set()
_RUNTIME_REDACTION_SECRETS_LOCK = threading.Lock()


@dataclass(frozen=True)
class CredentialResolution:
    """A credential value with a safe, non-secret representation."""

    value: str | None = field(repr=False)
    source: str

    @property
    def configured(self) -> bool:
        return self.value is not None


def register_redaction_secret(value: str | None) -> None:
    """Keep one secret redacted for the lifetime of this single-command CLI process."""

    if not isinstance(value, str) or len(value) < 4:
        return
    with _RUNTIME_REDACTION_SECRETS_LOCK:
        _RUNTIME_REDACTION_SECRETS.add(value)


def _runtime_redaction_secrets() -> tuple[str, ...]:
    with _RUNTIME_REDACTION_SECRETS_LOCK:
        return tuple(sorted(_RUNTIME_REDACTION_SECRETS, key=len, reverse=True))


def _redaction_secrets(env: Mapping[str, str] | None = None) -> tuple[str, ...]:
    """Return registered and configured secrets without exposing their values."""

    values = os.environ if env is None else env
    secrets = set(_runtime_redaction_secrets())
    for key in ("ESM_API_KEY", "HF_TOKEN", "MODAL_TOKEN_ID", "MODAL_TOKEN_SECRET"):
        secret = values.get(key, "")
        if secret and len(secret) >= 4:
            secrets.add(secret)
    return tuple(sorted(secrets, key=len, reverse=True))


def _crossing_secret_start(value: str, secrets: tuple[str, ...]) -> int | None:
    """Find the earliest known-secret prefix cut by the end of ``value``."""

    earliest: int | None = None
    for secret in secrets:
        maximum_visible = min(len(value), len(secret) - 1)
        if maximum_visible <= 0:
            continue
        search_start = len(value) - maximum_visible
        needle = secret[:4]
        position = value.find(needle, search_start)
        while position >= 0:
            visible = value[position:]
            if len(visible) < len(secret) and secret.startswith(visible):
                earliest = position if earliest is None else min(earliest, position)
                break
            position = value.find(needle, position + 1)
        for length in range(min(3, maximum_visible), 0, -1):
            if value.endswith(secret[:length]):
                position = len(value) - length
                earliest = position if earliest is None else min(earliest, position)
                break
    return earliest


def _bounded_secret_bytes(secret: str, limit: int) -> tuple[bytes, bool] | None:
    """Encode only enough of one secret to cover a bounded byte preview."""

    characters = secret[: limit + 1]
    try:
        encoded = characters.encode("utf-8", errors="surrogateescape")
    except UnicodeEncodeError:
        return None
    complete = len(characters) == len(secret) and len(encoded) <= limit + 1
    return encoded[: limit + 1], complete


def _bounded_secret_byte_forms(
    secrets: tuple[str, ...],
    *,
    limit: int,
) -> tuple[tuple[tuple[bytes, bool], ...], bool]:
    """Bound secret encodings and report values that have no safe byte form."""

    forms: list[tuple[bytes, bool]] = []
    has_unencodable = False
    for secret in secrets:
        form = _bounded_secret_bytes(secret, limit)
        if form is None:
            has_unencodable = True
        else:
            forms.append(form)
    return tuple(forms), has_unencodable


def _crossing_secret_bytes_start(
    value: bytes,
    secret_forms: tuple[tuple[bytes, bool], ...],
) -> int | None:
    """Find the earliest UTF-8 secret prefix cut by a byte-preview boundary."""

    earliest: int | None = None
    for prefix, complete in secret_forms:
        maximum_visible = len(value) if not complete else min(len(value), len(prefix) - 1)
        if maximum_visible <= 0 or not prefix:
            continue
        search_start = len(value) - maximum_visible
        needle = prefix[: min(4, len(prefix))]
        position = value.find(needle, search_start)
        while position >= 0:
            visible = value[position:]
            if prefix.startswith(visible) and (not complete or len(visible) < len(prefix)):
                earliest = position if earliest is None else min(earliest, position)
                break
            position = value.find(needle, position + 1)
        for length in range(min(3, maximum_visible, len(prefix)), 0, -1):
            if value.endswith(prefix[:length]):
                position = len(value) - length
                earliest = position if earliest is None else min(earliest, position)
                break
    return earliest


def _redact_complete_secret_bytes(
    value: bytes,
    secret_forms: tuple[tuple[bytes, bool], ...],
) -> bytes:
    """Replace complete known-secret byte spans before lossy UTF-8 decoding."""

    spans: list[tuple[int, int]] = []
    for secret, complete in secret_forms:
        if not complete or not secret:
            continue
        offset = 0
        while (position := value.find(secret, offset)) >= 0:
            spans.append((position, position + len(secret)))
            offset = position + len(secret)
    if not spans:
        return value

    merged: list[tuple[int, int]] = []
    for start, end in sorted(spans):
        if merged and start <= merged[-1][1]:
            merged[-1] = (merged[-1][0], max(merged[-1][1], end))
        else:
            merged.append((start, end))
    chunks: list[bytes] = []
    offset = 0
    for start, end in merged:
        chunks.extend((value[offset:start], b"[REDACTED]"))
        offset = end
    chunks.append(value[offset:])
    return b"".join(chunks)


CAMEL_CASE_BOUNDARY_RE = re.compile(r"(?<=[a-z0-9])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])")
BEARER_RE = re.compile(r"(?i)bearer\s+[A-Za-z0-9._~+/=-]+")
KEY_INITIAL_RE = r"[A-Za-z_\\]"
KEY_CHARACTER_RE = r"[-A-Za-z0-9_./\\]"
ASSIGNMENT_PREFIX_RE = re.compile(
    rf"""(?ix)
    (?<![A-Za-z0-9_\\-])
    (?:
      (?P<bracket_owner>{KEY_INITIAL_RE}{KEY_CHARACTER_RE}*)\s*\[\s*
      (?:
        (?P<bracket_quote>(?:\\*["']))
        (?P<bracket_quoted_key>{KEY_INITIAL_RE}(?:{KEY_CHARACTER_RE}|\s)*?)
        (?P=bracket_quote)
        |
        (?P<bracket_bare_key>{KEY_INITIAL_RE}{KEY_CHARACTER_RE}*)
        |
        (?P<bracket_index>[0-9]+)
      )\s*\]
      |
      (?P<key_quote>(?:\\*["']))
      (?P<quoted_key>{KEY_INITIAL_RE}(?:{KEY_CHARACTER_RE}|\s)*?)
      (?P=key_quote)
      |
      (?P<spaced_key>
        (?:secret\s+access\s+key|authorization\s+header|access\s+key(?:\s+id)?
        |api\s+key|private\s+key|client\s+secret|client\s+credentials?
        |secret\s+key|token\s+secret|password\s+hash)
      )
      |
      (?P<bare_key>{KEY_INITIAL_RE}{KEY_CHARACTER_RE}*)
    )
    (?P<separator>\s*(?:[=:]|%3d)\s*)
    """
)
URL_SECRET_PARAM_RE = re.compile(
    r"(?i)([?&](?:x-amz-credential|x-amz-signature|x-amz-security-token|token|credential|signature)=)[^&#\s]+"
)


def _configured(env: Mapping[str, str], key: str) -> bool:
    return bool(env.get(key, "").strip())


def missing_esm_api_key_message() -> str:
    """The one message shown when managed access has no credential.

    Both the transport and the ESMC mutation command raise this, so the
    acquisition URL cannot drift between them.
    """

    return (
        "ESM_API_KEY is missing. Get a free key at "
        f"{BIOHUB_API_KEY_URL} (existing Forge accounts sign in with the same "
        "credentials), set it as ESM_API_KEY in the environment that launches "
        "Codex, then restart Codex."
    )


def resolve_esm_api_key(
    env: Mapping[str, str] | None = None,
    *,
    system_name: str | None = None,
    keychain_run: Callable[..., Any] | None = None,
) -> CredentialResolution:
    """Resolve Biohub auth without displaying, logging, or persisting the value.

    ``ESM_API_KEY`` always wins. On macOS only, an absent environment value may
    fall back to the verified generic-password service used by the Codex host.
    All lookup failures are intentionally indistinguishable from a missing key.
    """

    values = os.environ if env is None else env
    environment_value = values.get("ESM_API_KEY", "").strip()
    if environment_value:
        return CredentialResolution(value=environment_value, source="environment")

    if values.get(DISABLE_KEYCHAIN_ENV, "").strip() == "1":
        return CredentialResolution(value=None, source="none")

    if (platform.system() if system_name is None else system_name) != "Darwin":
        return CredentialResolution(value=None, source="none")

    runner = subprocess.run if keychain_run is None else keychain_run
    try:
        completed = runner(
            [
                "/usr/bin/security",
                "find-generic-password",
                "-s",
                MACOS_KEYCHAIN_SERVICE,
                "-w",
            ],
            stdin=subprocess.DEVNULL,
            stdout=subprocess.PIPE,
            stderr=subprocess.PIPE,
            text=True,
            check=False,
            shell=False,
            timeout=KEYCHAIN_LOOKUP_TIMEOUT_SECONDS,
        )
    except (OSError, subprocess.SubprocessError):
        return CredentialResolution(value=None, source="none")

    if completed.returncode != 0:
        return CredentialResolution(value=None, source="none")
    keychain_value = completed.stdout.strip()
    if not keychain_value:
        return CredentialResolution(value=None, source="none")
    return CredentialResolution(value=keychain_value, source="macos-keychain")


def _macos_keychain_item_is_configured(
    *,
    system_name: str | None = None,
    keychain_run: Callable[..., Any] | None = None,
) -> bool:
    """Probe item presence without requesting or materializing its secret value."""

    if (platform.system() if system_name is None else system_name) != "Darwin":
        return False
    runner = subprocess.run if keychain_run is None else keychain_run
    try:
        completed = runner(
            [
                "/usr/bin/security",
                "find-generic-password",
                "-s",
                MACOS_KEYCHAIN_SERVICE,
            ],
            stdin=subprocess.DEVNULL,
            stdout=subprocess.DEVNULL,
            stderr=subprocess.DEVNULL,
            check=False,
            shell=False,
            timeout=KEYCHAIN_LOOKUP_TIMEOUT_SECONDS,
        )
    except (OSError, subprocess.SubprocessError):
        return False
    return completed.returncode == 0


SENSITIVE_KEY_SEGMENTS = {
    "authorization",
    "credential",
    "credentials",
    "password",
    "passwords",
    "secret",
    "secrets",
    "signature",
    "signatures",
}
SAFE_SECRET_METADATA = {"count", "counts", "length", "lengths"}
SAFE_TOKEN_METADATA = {
    "activation",
    "activations",
    "attention",
    "attentions",
    "count",
    "counts",
    "dropout",
    "embedding",
    "embeddings",
    "entropy",
    "entropies",
    "expires",
    "expiration",
    "expirations",
    "expiry",
    "feature",
    "features",
    "hidden",
    "ids",
    "indices",
    "indexes",
    "length",
    "lengths",
    "logit",
    "logits",
    "mask",
    "masks",
    "offset",
    "offsets",
    "position",
    "positions",
    "probabilities",
    "probability",
    "representation",
    "representations",
    "saliency",
    "score",
    "scores",
    "state",
    "states",
    "type",
    "types",
    "values",
    "vector",
    "vectors",
}
SAFE_PLURAL_TOKEN_PREFIXES = {
    "acid",
    "added",
    "amino",
    "cached",
    "completion",
    "input",
    "max",
    "min",
    "model",
    "num",
    "output",
    "overflowing",
    "prompt",
    "protein",
    "reasoning",
    "residue",
    "sae",
    "sequence",
    "special",
    "suppress",
    "total",
    "truncated",
    "vocab",
}
TOKENIZER_TOKEN_PREFIXES = {
    "bos",
    "cls",
    "eos",
    "mask",
    "pad",
    "sep",
    "unk",
}
SAFE_EXACT_SCIENTIFIC_KEYS = {"token_to_atoms"}
SAFE_COMPACT_TOKENIZER_KEYS = {
    f"{modifier}{prefix}token{suffix}"
    for modifier in ("", "forced", "tokenizer", "tokenizerforced")
    for prefix in TOKENIZER_TOKEN_PREFIXES
    for suffix in ("", "id", "ids")
} | {
    "addedtokensdecoder",
    "decoderstarttoken",
    "decoderstarttokenid",
    "decoderstarttokenids",
    "numtruncatedtokens",
    "specialtokensmask",
}
SAFE_SCIENTIFIC_TOKEN_PREFIXES = SAFE_PLURAL_TOKEN_PREFIXES | {
    "batch",
    "chain",
    "layer",
    "per",
    "position",
}
SAFE_COMPACT_SCIENTIFIC_TOKEN_PREFIXES = SAFE_SCIENTIFIC_TOKEN_PREFIXES | {
    "aminoacid",
    "perposition",
    "perresidue",
}
DYNAMIC_SCIENTIFIC_SCOPE_HEADS = {"layer", "position", "residue"}
DYNAMIC_COMPACT_SCIENTIFIC_PREFIX_RE = re.compile(r"(?:sae|per)?(?:layer|position|residue)[0-9]+")
EXPLICIT_CHAIN_SNAKE_RE = re.compile(
    r"^(?P<lead>(?:[a-z]+_)*)chain_[A-Z0-9]{1,4}_token_(?P<metadata>[a-z0-9_]+)$"
)
EXPLICIT_CHAIN_CAMEL_RE = re.compile(
    r"^(?:(?P<lead>[a-z]+)Chain|chain)[A-Z0-9]{1,4}Token(?P<metadata>[A-Z][A-Za-z0-9]*)$"
)
CREDENTIAL_TOKEN_QUALIFIERS = {
    "access",
    "api",
    "auth",
    "authentication",
    "aws",
    "bearer",
    "biohub",
    "csrf",
    "esm",
    "gcp",
    "gh",
    "github",
    "gitlab",
    "hf",
    "huggingface",
    "idp",
    "jwt",
    "modal",
    "oauth",
    "oauth2",
    "oidc",
    "okta",
    "openai",
    "pat",
    "refresh",
    "security",
    "session",
    "slack",
    "sso",
    "sts",
}
SAFE_CREDENTIAL_TOKEN_METADATA = {
    "count",
    "counts",
    "expires",
    "expiration",
    "expirations",
    "expiry",
    "length",
    "lengths",
    "type",
    "types",
}
SAFE_COMPACT_CREDENTIAL_TOKEN_METADATA = {
    "count",
    "counts",
    "expires",
    "expiresat",
    "expiration",
    "expirations",
    "expiry",
    "length",
    "lengths",
    "type",
    "types",
}
COMPACT_SECRET_SUFFIXES = {
    "accesskey",
    "accesskeyid",
    "apikey",
    "apikeys",
    "authorizationheader",
    "clientcredentials",
    "clientsecret",
    "clientsecrets",
    "credentials",
    "passwordhash",
    "privatekey",
    "secretaccesskey",
    "secretkey",
    "token",
    "tokenid",
    "tokensecret",
}


def _key_segments(key: object) -> tuple[str, ...]:
    normalized = CAMEL_CASE_BOUNDARY_RE.sub("_", str(key))
    return tuple(segment.lower() for segment in re.split(r"[^A-Za-z0-9]+", normalized) if segment)


def _decode_key_label(key: str) -> str:
    def replace(match: re.Match[str]) -> str:
        return chr(int(match.group(1), 16))

    decoded = re.sub(r"\\+u([0-9a-fA-F]{4})", replace, key)
    return re.sub(r"\\+x([0-9a-fA-F]{2})", replace, decoded)


def _assignment_keys(match: re.Match[str]) -> tuple[str, ...]:
    bracket_owner = match.group("bracket_owner")
    if bracket_owner is not None:
        child = match.group("bracket_quoted_key") or match.group("bracket_bare_key")
        keys = [bracket_owner]
        if child is not None:
            keys.append(child)
        return tuple(_decode_key_label(key) for key in keys)
    key = match.group("quoted_key") or match.group("spaced_key") or match.group("bare_key")
    return (_decode_key_label(key),)


def _is_scientific_token_prefix(prefix: tuple[str, ...]) -> bool:
    for index, item in enumerate(prefix):
        if item in SAFE_SCIENTIFIC_TOKEN_PREFIXES:
            continue
        previous = prefix[index - 1] if index else None
        if item.isdigit() and previous in DYNAMIC_SCIENTIFIC_SCOPE_HEADS:
            continue
        if re.fullmatch(r"(?:layer|position|residue)[0-9]+", item):
            continue
        return False
    return True


def _is_compact_credential_token_prefix(prefix: str) -> bool:
    return any(prefix.endswith(qualifier) for qualifier in CREDENTIAL_TOKEN_QUALIFIERS)


def _is_compact_scientific_token_prefix(prefix: str) -> bool:
    return prefix in SAFE_COMPACT_SCIENTIFIC_TOKEN_PREFIXES or bool(
        DYNAMIC_COMPACT_SCIENTIFIC_PREFIX_RE.fullmatch(prefix)
    )


def _has_explicit_scientific_chain_scope(key: str) -> bool:
    snake = EXPLICIT_CHAIN_SNAKE_RE.fullmatch(key)
    if snake is not None:
        lead = tuple(item for item in snake.group("lead").split("_") if item)
        metadata = _key_segments(snake.group("metadata"))
        return (
            _is_scientific_token_prefix(lead)
            and bool(metadata)
            and all(item in SAFE_TOKEN_METADATA for item in metadata)
        )
    camel = EXPLICIT_CHAIN_CAMEL_RE.fullmatch(key)
    if camel is None:
        return False
    lead = camel.group("lead")
    metadata = _key_segments(camel.group("metadata"))
    return (
        (lead is None or lead in SAFE_SCIENTIFIC_TOKEN_PREFIXES)
        and bool(metadata)
        and all(item in SAFE_TOKEN_METADATA for item in metadata)
    )


def _is_secret_key(key: object) -> bool:
    """Recognize credential labels while preserving model-token metadata."""

    raw_key = str(key)
    if raw_key in SAFE_EXACT_SCIENTIFIC_KEYS:
        return False
    explicit_chain_scope = _has_explicit_scientific_chain_scope(raw_key)
    segments = _key_segments(raw_key)
    for index, segment in enumerate(segments):
        if segment not in SENSITIVE_KEY_SEGMENTS:
            continue
        tail = segments[index + 1 :]
        if segment == "secret" and tail and tail[0] in SAFE_SECRET_METADATA:
            continue
        return True

    for left, right in itertools.pairwise(segments):
        if left in {"access", "api", "private"} and right in {"key", "keys"}:
            return True

    for index, segment in enumerate(segments):
        if segment == "token":
            prefix = segments[:index]
            tail = segments[index + 1 :]
            credential_qualified = any(item in CREDENTIAL_TOKEN_QUALIFIERS for item in prefix)
            if credential_qualified:
                if tail and tail[0] in SAFE_CREDENTIAL_TOKEN_METADATA:
                    continue
                return True
            tokenizer_prefix = bool(
                prefix
                and (prefix[-1] in TOKENIZER_TOKEN_PREFIXES or prefix[-2:] == ("decoder", "start"))
            )
            if tokenizer_prefix and (not tail or tail[0] in {"id", "ids"}):
                continue
            unknown_prefix = bool(prefix) and not (
                explicit_chain_scope or _is_scientific_token_prefix(prefix)
            )
            if unknown_prefix:
                if tail and tail[0] in SAFE_CREDENTIAL_TOKEN_METADATA:
                    continue
                return True
            if not tail or tail[0] not in SAFE_TOKEN_METADATA:
                return True
        elif segment == "tokens":
            prefix = segments[index - 1] if index else None
            if prefix is not None and prefix not in SAFE_PLURAL_TOKEN_PREFIXES:
                return True

    compact = re.sub(r"[^a-z0-9]", "", str(key).lower())
    if compact in SAFE_COMPACT_TOKENIZER_KEYS:
        return False
    for metadata in sorted(SAFE_TOKEN_METADATA, key=len, reverse=True):
        marker = f"token{metadata}"
        if not compact.endswith(marker):
            continue
        prefix = compact[: -len(marker)]
        if prefix and metadata not in SAFE_COMPACT_CREDENTIAL_TOKEN_METADATA:
            if not explicit_chain_scope and _is_compact_credential_token_prefix(prefix):
                return True
            if not (explicit_chain_scope or _is_compact_scientific_token_prefix(prefix)):
                return True
        break
    if any(compact.endswith(suffix) for suffix in COMPACT_SECRET_SUFFIXES):
        return True
    return False


def _quote_wrapper(value: str, start: int) -> str | None:
    index = start
    while index < len(value) and value[index] == "\\":
        index += 1
    if index < len(value) and value[index] in {'"', "'"}:
        return value[start : index + 1]
    return None


def _quoted_value_end(value: str, start: int, wrapper: str) -> int:
    quote = wrapper[-1]
    wrapper_backslashes = len(wrapper) - 1
    search = start + len(wrapper)
    while True:
        closing = value.find(quote, search)
        if closing < 0:
            return len(value)
        backslashes = 0
        cursor = closing - 1
        while cursor >= start and value[cursor] == "\\":
            backslashes += 1
            cursor -= 1
        wrapper_matches = (
            backslashes % 2 == 0 if wrapper_backslashes == 0 else backslashes == wrapper_backslashes
        )
        end = closing + 1
        if wrapper_matches and (end == len(value) or value[end] in " \t\r\n,;}]"):
            return end
        search = closing + 1


def _quoted_segment_end(value: str, start: int, *, stop_at_record_delimiter: bool) -> int | None:
    quote = value[start]
    escaped = False
    for index in range(start + 1, len(value)):
        character = value[index]
        if escaped:
            escaped = False
        elif character == "\\":
            escaped = True
        elif character == quote:
            return index + 1
        elif stop_at_record_delimiter and character in "\r\n,;":
            return None
    return None


def _advance_quote_context(
    value: str,
    start: int,
    end: int,
    *,
    quote: str | None,
    escaped: bool,
) -> tuple[str | None, bool]:
    """Scan quote state incrementally without one pointer per input character."""

    for index in range(start, end):
        character = value[index]
        if quote is not None:
            if escaped:
                escaped = False
            elif character == "\\":
                escaped = True
            elif character == quote:
                quote = None
        elif character in {'"', "'"} and (index == 0 or value[index - 1] in " \t\r\n=:[{,("):
            quote = character
    return quote, escaped


def _unquoted_value_end(
    value: str,
    start: int,
    *,
    allow_spaces: bool,
    outer_quote: str | None,
) -> int:
    pairs = {"(": ")", "[": "]", "{": "}"}
    stack: list[str] = []
    index = start
    while index < len(value):
        if value.startswith("[REDACTED]", index):
            marker_end = index + len("[REDACTED]")
            if not stack and (marker_end == len(value) or value[marker_end] in " \t\r\n,;}]\"'"):
                return marker_end
            index = marker_end
            continue

        character = value[index]
        if character in {'"', "'"}:
            quoted_end = _quoted_segment_end(
                value,
                index,
                stop_at_record_delimiter=not stack,
            )
            if quoted_end is None:
                if not stack and character == outer_quote:
                    return index
                return len(value)
            index = quoted_end
            continue
        if character in pairs:
            stack.append(pairs[character])
            index += 1
            continue
        if stack and character == stack[-1]:
            stack.pop()
            index += 1
            continue
        if not stack and character in "\r\n,;}]":
            return index
        if not stack and character.isspace():
            if not allow_spaces:
                return index
            next_start = index
            while next_start < len(value) and value[next_start].isspace():
                next_start += 1
            if ASSIGNMENT_PREFIX_RE.match(value, next_start):
                return index
        index += 1
    return len(value)


def _secret_value_end_and_replacement(
    value: str,
    start: int,
    *,
    key: str,
    outer_quote: str | None,
) -> tuple[int, str]:
    if start >= len(value):
        return start, "[REDACTED]"
    wrapper = _quote_wrapper(value, start)
    if wrapper is not None:
        return (
            _quoted_value_end(value, start, wrapper),
            f"{wrapper}[REDACTED]{wrapper}",
        )
    return (
        _unquoted_value_end(
            value,
            start,
            allow_spaces="authorization" in _key_segments(key),
            outer_quote=outer_quote,
        ),
        "[REDACTED]",
    )


def _redact_secret_assignments(value: str) -> str:
    pieces: list[str] = []
    cursor = 0
    search = 0
    context_cursor = 0
    quote: str | None = None
    escaped = False
    while match := ASSIGNMENT_PREFIX_RE.search(value, search):
        quote, escaped = _advance_quote_context(
            value,
            context_cursor,
            match.start(),
            quote=quote,
            escaped=escaped,
        )
        context_cursor = match.start()
        keys = _assignment_keys(match)
        if not any(_is_secret_key(key) for key in keys):
            search = match.start() + 1
            continue
        end, replacement = _secret_value_end_and_replacement(
            value,
            match.end(),
            key=" ".join(keys),
            outer_quote=quote,
        )
        pieces.extend((value[cursor : match.end()], replacement))
        cursor = end
        search = max(end, match.end())
    pieces.append(value[cursor:])
    return "".join(pieces)


def credential_preflight(
    env: Mapping[str, str] | None = None,
    *,
    home: Path | None = None,
    system_name: str | None = None,
    keychain_run: Callable[..., Any] | None = None,
) -> dict[str, object]:
    values = os.environ if env is None else env
    root = Path.home() if home is None else home
    biohub_environment = _configured(values, "ESM_API_KEY")
    biohub_keychain = (
        not biohub_environment
        and values.get(DISABLE_KEYCHAIN_ENV, "").strip() != "1"
        and _macos_keychain_item_is_configured(
            system_name=system_name,
            keychain_run=keychain_run,
        )
    )
    modal_env = _configured(values, "MODAL_TOKEN_ID") and _configured(values, "MODAL_TOKEN_SECRET")
    modal_profile = (root / ".modal.toml").is_file()
    biohub_configured = biohub_environment or biohub_keychain
    biohub_managed: dict[str, str] = {
        "status": "configured" if biohub_configured else "missing",
        "source": (
            "environment"
            if biohub_environment
            else ("macos-keychain" if biohub_keychain else "none")
        ),
    }
    if not biohub_configured:
        # Reporting only "missing" leaves the caller nothing to act on, so the
        # variable to set and the page that issues a key travel with the status.
        biohub_managed["variable"] = "ESM_API_KEY"
        biohub_managed["obtain_key_url"] = BIOHUB_API_KEY_URL
        biohub_managed["quickstart_url"] = BIOHUB_QUICKSTART_URL
    return {
        "biohub_managed": biohub_managed,
        "atlas": {"status": "not-required", "source": "public v1alpha1 API and anonymous S3"},
        "modal": {
            "status": "configured" if modal_env or modal_profile else "missing",
            "source": "environment" if modal_env else ("profile" if modal_profile else "none"),
        },
        "hugging_face": {
            "status": "configured" if _configured(values, "HF_TOKEN") else "optional-missing",
            "source": "HF_TOKEN (optional for public weights)",
        },
    }


def redact_text(value: str, env: Mapping[str, str] | None = None) -> str:
    result = BEARER_RE.sub("Bearer [REDACTED]", value)
    result = URL_SECRET_PARAM_RE.sub(lambda match: f"{match.group(1)}[REDACTED]", result)
    result = _redact_secret_assignments(result)
    for secret in _redaction_secrets(env):
        result = result.replace(secret, "[REDACTED]")
    return result


def redact_truncated_text(
    value: str,
    *,
    truncated: bool,
    env: Mapping[str, str] | None = None,
) -> str:
    """Redact a bounded text prefix without leaking a cut known secret."""

    if truncated:
        crossing = _crossing_secret_start(value, _redaction_secrets(env))
        if crossing is not None:
            value = value[:crossing] + "[REDACTED]"
    return redact_text(value, env)


def redact_bounded_bytes_preview(
    value: bytes,
    *,
    limit: int,
    env: Mapping[str, str] | None = None,
) -> str:
    """Return a redacted UTF-8 preview without copying an unbounded byte body."""

    preview = value[:limit]
    truncated = len(value) > len(preview)
    secret_forms, has_unencodable_secret = _bounded_secret_byte_forms(
        _redaction_secrets(env),
        limit=limit,
    )
    if truncated and has_unencodable_secret:
        return "[REDACTED]"
    suffix = ""
    if truncated:
        crossing = _crossing_secret_bytes_start(preview, secret_forms)
        if crossing is not None:
            preview = preview[:crossing]
            suffix = "[REDACTED]"
    preview = _redact_complete_secret_bytes(preview, secret_forms)
    return redact_text(preview.decode("utf-8", errors="replace"), env) + suffix


def redact_mapping_key(value: object, env: Mapping[str, str] | None = None) -> str:
    """Redact credential material embedded in a mapping key."""

    return redact_text(str(value), env)


def _unique_mapping_key(candidate: str, existing: Mapping[str, Any], index: int) -> str:
    if candidate not in existing:
        return candidate
    suffix = 1
    while True:
        resolved = f"{candidate} [duplicate-{index}-{suffix}]"
        if resolved not in existing:
            return resolved
        suffix += 1


def redact(value: Any, env: Mapping[str, str] | None = None) -> Any:
    if isinstance(value, Mapping):
        result: dict[str, Any] = {}
        for index, (key, item) in enumerate(value.items()):
            safe_key = _unique_mapping_key(redact_mapping_key(key, env), result, index)
            result[safe_key] = "[REDACTED]" if _is_secret_key(key) else redact(item, env)
        return result
    if isinstance(value, list):
        return [redact(item, env) for item in value]
    if isinstance(value, tuple):
        return tuple(redact(item, env) for item in value)
    if isinstance(value, str):
        return redact_text(value, env)
    return value

SHA-256: b6515ba737b5b105f4b0ced872d72db6390a22ebff2eeed43750347c4297bc5c