← Files Auto Preference LearnerARCHIVED FILE
skills/auto-preference-learner/scripts/_common.py
5.87 KB · Oct 2, 2026 · 00:29 UTC
from __future__ import annotations
import hashlib
import json
import os
import subprocess
import tempfile
from datetime import datetime, timezone
from functools import lru_cache
from pathlib import Path
from typing import Any, Iterable, Iterator
START_MARKER = "<!-- auto-preference-learner:start -->"
END_MARKER = "<!-- auto-preference-learner:end -->"
STATE_DIR_NAME = "auto-preference-learner"
def codex_home(value: str | None = None) -> Path:
raw = value or os.environ.get("CODEX_HOME")
return Path(raw).expanduser().resolve() if raw else (Path.home() / ".codex").resolve()
def utc_now() -> datetime:
return datetime.now(timezone.utc)
def isoformat(value: datetime | None = None) -> str:
return (value or utc_now()).astimezone(timezone.utc).isoformat().replace("+00:00", "Z")
def parse_timestamp(value: Any) -> datetime | None:
if isinstance(value, (int, float)):
try:
return datetime.fromtimestamp(float(value), tz=timezone.utc)
except (OverflowError, OSError, ValueError):
return None
if not isinstance(value, str) or not value.strip():
return None
text = value.strip()
if text.endswith("Z"):
text = text[:-1] + "+00:00"
try:
parsed = datetime.fromisoformat(text)
except ValueError:
return None
if parsed.tzinfo is None:
parsed = parsed.replace(tzinfo=timezone.utc)
return parsed.astimezone(timezone.utc)
def read_jsonl(path: Path) -> Iterator[tuple[int, dict[str, Any]]]:
with path.open("r", encoding="utf-8-sig", errors="replace") as handle:
for line_number, line in enumerate(handle, 1):
if not line.strip():
continue
try:
value = json.loads(line)
except json.JSONDecodeError:
continue
if isinstance(value, dict):
yield line_number, value
def extract_text(value: Any) -> str:
if isinstance(value, str):
return value.strip()
if isinstance(value, list):
parts: list[str] = []
for item in value:
if isinstance(item, str):
parts.append(item)
elif isinstance(item, dict):
kind = item.get("type")
if kind in {"text", "input_text", "output_text"} or "text" in item:
text = item.get("text")
if isinstance(text, str):
parts.append(text)
return "\n".join(part.strip() for part in parts if part.strip()).strip()
if isinstance(value, dict):
for key in ("text", "message", "content"):
text = extract_text(value.get(key))
if text:
return text
return ""
def sha256_bytes(value: bytes) -> str:
return hashlib.sha256(value).hexdigest()
def stable_id(*parts: str, prefix: str = "") -> str:
digest = hashlib.sha256("\0".join(parts).encode("utf-8")).hexdigest()[:16]
return f"{prefix}{digest}"
def load_json(path: Path, default: Any) -> Any:
if not path.exists():
return default
with path.open("r", encoding="utf-8-sig") as handle:
return json.load(handle)
def atomic_write_text(path: Path, content: str) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
descriptor, temporary = tempfile.mkstemp(prefix=f".{path.name}.", suffix=".tmp", dir=path.parent)
try:
with os.fdopen(descriptor, "w", encoding="utf-8", newline="") as handle:
handle.write(content)
handle.flush()
os.fsync(handle.fileno())
os.replace(temporary, path)
except BaseException:
try:
os.unlink(temporary)
except FileNotFoundError:
pass
raise
def atomic_write_bytes(path: Path, content: bytes) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
descriptor, temporary = tempfile.mkstemp(prefix=f".{path.name}.", suffix=".tmp", dir=path.parent)
try:
with os.fdopen(descriptor, "wb") as handle:
handle.write(content)
handle.flush()
os.fsync(handle.fileno())
os.replace(temporary, path)
except BaseException:
try:
os.unlink(temporary)
except FileNotFoundError:
pass
raise
def atomic_write_json(path: Path, value: Any) -> None:
atomic_write_text(path, json.dumps(value, ensure_ascii=False, indent=2) + "\n")
def write_json_output(value: Any, output: str | None) -> None:
text = json.dumps(value, ensure_ascii=False, indent=2) + "\n"
if output:
atomic_write_text(Path(output).expanduser().resolve(), text)
else:
print(text, end="")
def find_git_root(cwd: Path) -> Path | None:
if not cwd.is_dir():
return None
result = subprocess.run(
["git", "-C", str(cwd), "rev-parse", "--show-toplevel"],
capture_output=True,
text=True,
encoding="utf-8",
errors="replace",
check=False,
)
if result.returncode != 0 or not result.stdout.strip():
return None
return Path(result.stdout.strip()).resolve()
@lru_cache(maxsize=4096)
def project_root_for_cwd(value: str | None) -> Path | None:
if not value:
return None
try:
cwd = Path(value).expanduser().resolve()
except (OSError, RuntimeError):
return None
return find_git_root(cwd) or (cwd if cwd.is_dir() else None)
def normalize_text(value: str) -> str:
return " ".join(value.split()).casefold()
def iter_dicts(value: Any) -> Iterable[dict[str, Any]]:
if isinstance(value, dict):
yield value
for child in value.values():
yield from iter_dicts(child)
elif isinstance(value, list):
for child in value:
yield from iter_dicts(child)
SHA-256: 05b118b177936ddcddaa4061103631b341bc6d4df51d182ee55bce30513aaf9d