← Files Empire LLM for CodexARCHIVED FILE

scripts/comparative_utility.py

10.4 KB · Oct 2, 2026 · 00:29 UTC

↓ Download file

#!/usr/bin/env python3
"""Prepare and score privacy-preserving blinded Empire utility evaluations."""

from __future__ import annotations

import argparse
import hashlib
import json
from pathlib import Path
from typing import Any


class EvaluationError(Exception):
    pass


ROUTES = ("codex_alone", "codex_plus_empire")


def load_json(path: Path) -> dict[str, Any]:
    try:
        value = json.loads(path.read_text(encoding="utf-8"))
    except (OSError, json.JSONDecodeError) as exc:
        raise EvaluationError(f"Unable to read evaluation JSON: {path}") from exc
    if not isinstance(value, dict):
        raise EvaluationError("Evaluation input must be a JSON object")
    return value


def write_json(path: Path, value: dict[str, Any]) -> None:
    path.parent.mkdir(parents=True, exist_ok=True)
    temporary = path.with_suffix(path.suffix + ".tmp")
    temporary.write_text(json.dumps(value, indent=2) + "\n", encoding="utf-8")
    temporary.replace(path)


def candidate_id(seed: str, task_id: str, response_hash: str) -> str:
    digest = hashlib.sha256(
        f"{seed}\0{task_id}\0{response_hash}".encode("utf-8")
    ).hexdigest()[:20]
    return f"candidate-{digest}"


def validate_plan(plan: dict[str, Any]) -> tuple[list[str], dict[str, float]]:
    tasks = plan.get("tasks")
    thresholds = plan.get("thresholds")
    if plan.get("schema_version") != 1 or not isinstance(tasks, list):
        raise EvaluationError("Evaluation plan schema is invalid")
    raw_task_ids = [item.get("id") for item in tasks if isinstance(item, dict)]
    if len(raw_task_ids) < 5 or any(
        not isinstance(item, str) or not item for item in raw_task_ids
    ):
        raise EvaluationError("Evaluation plan requires at least five named tasks")
    task_ids = [item for item in raw_task_ids if isinstance(item, str)]
    if len(set(task_ids)) != len(task_ids):
        raise EvaluationError("Evaluation task IDs must be unique")
    if not isinstance(thresholds, dict):
        raise EvaluationError("Evaluation thresholds are missing")
    required = (
        "minimum_verified_finding_gain",
        "maximum_false_positive_delta",
        "minimum_actionable_rate",
        "maximum_cost_per_task_usd",
    )
    parsed: dict[str, float] = {}
    for key in required:
        value = thresholds.get(key)
        if not isinstance(value, (int, float)) or isinstance(value, bool):
            raise EvaluationError(f"Invalid threshold: {key}")
        parsed[key] = float(value)
    return task_ids, parsed


def prepare(
    plan: dict[str, Any], runs: dict[str, Any]
) -> tuple[dict[str, Any], dict[str, Any]]:
    task_ids, _ = validate_plan(plan)
    seed = runs.get("blinding_seed")
    records = runs.get("runs")
    if not isinstance(seed, str) or len(seed) < 16 or not isinstance(records, list):
        raise EvaluationError("Run manifest requires a private blinding seed and runs")
    by_task: dict[str, dict[str, dict[str, Any]]] = {}
    for item in records:
        if not isinstance(item, dict):
            raise EvaluationError("Run entries must be objects")
        task_id = item.get("task_id")
        route = item.get("route")
        response_hash = item.get("response_hash")
        if task_id not in task_ids or route not in ROUTES:
            raise EvaluationError("Run entry has an unknown task or route")
        if not isinstance(response_hash, str) or not response_hash.startswith("sha256:"):
            raise EvaluationError("Run response hashes must be SHA-256 references")
        by_task.setdefault(task_id, {})[route] = item
    if any(set(by_task.get(task_id, {})) != set(ROUTES) for task_id in task_ids):
        raise EvaluationError("Every task requires both evaluation routes")

    candidates: list[dict[str, Any]] = []
    mapping: list[dict[str, Any]] = []
    for task_id in task_ids:
        for route in ROUTES:
            item = by_task[task_id][route]
            opaque = candidate_id(seed, task_id, item["response_hash"])
            candidates.append(
                {
                    "candidate_id": opaque,
                    "task_id": task_id,
                    "response_hash": item["response_hash"],
                    "response_artifact": f"{opaque}.md",
                }
            )
            mapping.append(
                {
                    "candidate_id": opaque,
                    "task_id": task_id,
                    "route": route,
                    "latency_ms": item.get("latency_ms"),
                    "cost_usd": item.get("cost_usd"),
                    "source_response_artifact": item.get("response_artifact"),
                }
            )
    candidates.sort(key=lambda item: item["candidate_id"])
    return (
        {
            "schema_version": 1,
            "evaluation_id": plan.get("evaluation_id"),
            "blind": True,
            "candidate_count": len(candidates),
            "candidates": candidates,
            "instructions": plan.get("adjudication_instructions"),
        },
        {
            "schema_version": 1,
            "evaluation_id": plan.get("evaluation_id"),
            "private": True,
            "mapping": mapping,
        },
    )


def _average(values: list[float]) -> float:
    return sum(values) / len(values) if values else 0.0


def score(
    plan: dict[str, Any], mapping: dict[str, Any], judgments: dict[str, Any]
) -> dict[str, Any]:
    task_ids, thresholds = validate_plan(plan)
    if mapping.get("private") is not True or judgments.get("blind") is not True:
        raise EvaluationError("Scoring requires a private mapping and blind judgments")
    if judgments.get("independent_adjudicator") is not True:
        raise EvaluationError("The adjudicator must be independent of production routing")
    map_rows = mapping.get("mapping")
    judgment_rows = judgments.get("judgments")
    if not isinstance(map_rows, list) or not isinstance(judgment_rows, list):
        raise EvaluationError("Mapping or judgments are missing")
    mapped = {
        row.get("candidate_id"): row
        for row in map_rows
        if isinstance(row, dict) and row.get("route") in ROUTES
    }
    judged = {
        row.get("candidate_id"): row
        for row in judgment_rows
        if isinstance(row, dict)
    }
    if set(mapped) != set(judged) or len(mapped) != len(task_ids) * 2:
        raise EvaluationError("Every blinded candidate must be judged exactly once")

    aggregates: dict[str, dict[str, Any]] = {}
    for route in ROUTES:
        route_rows = [
            (mapped[candidate], judged[candidate])
            for candidate in mapped
            if mapped[candidate]["route"] == route
        ]
        verified = [float(judgment.get("verified_findings", 0)) for _, judgment in route_rows]
        false_positives = [
            float(judgment.get("false_positives", 0)) for _, judgment in route_rows
        ]
        actionable = [1.0 if judgment.get("actionable") is True else 0.0 for _, judgment in route_rows]
        costs = [float(row.get("cost_usd") or 0) for row, _ in route_rows]
        latencies = [float(row.get("latency_ms") or 0) for row, _ in route_rows]
        aggregates[route] = {
            "task_count": len(route_rows),
            "mean_verified_findings": round(_average(verified), 4),
            "mean_false_positives": round(_average(false_positives), 4),
            "actionable_rate": round(_average(actionable), 4),
            "mean_cost_usd": round(_average(costs), 6),
            "mean_latency_ms": round(_average(latencies), 2),
        }

    baseline = aggregates["codex_alone"]
    empire = aggregates["codex_plus_empire"]
    verified_gain = round(
        empire["mean_verified_findings"] - baseline["mean_verified_findings"], 4
    )
    false_positive_delta = round(
        empire["mean_false_positives"] - baseline["mean_false_positives"], 4
    )
    checks = {
        "complete_task_set": all(
            aggregates[route]["task_count"] == len(task_ids) for route in ROUTES
        ),
        "verified_finding_gain": verified_gain
        >= thresholds["minimum_verified_finding_gain"],
        "false_positive_delta": false_positive_delta
        <= thresholds["maximum_false_positive_delta"],
        "actionable_rate": empire["actionable_rate"]
        >= thresholds["minimum_actionable_rate"],
        "cost_ceiling": empire["mean_cost_usd"]
        <= thresholds["maximum_cost_per_task_usd"],
    }
    passed = all(checks.values())
    return {
        "schema_version": 1,
        "evaluation_id": plan.get("evaluation_id"),
        "executed_at": judgments.get("executed_at"),
        "blind": True,
        "independent_adjudicator": True,
        "task_count": len(task_ids),
        "routes": aggregates,
        "verified_finding_gain": verified_gain,
        "false_positive_delta": false_positive_delta,
        "thresholds": thresholds,
        "checks": checks,
        "raw_prompts_persisted": False,
        "raw_responses_persisted": False,
        "passed": passed,
    }


def main() -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    sub = parser.add_subparsers(dest="command", required=True)
    prepare_parser = sub.add_parser("prepare")
    prepare_parser.add_argument("--plan", type=Path, required=True)
    prepare_parser.add_argument("--runs", type=Path, required=True)
    prepare_parser.add_argument("--blind-output", type=Path, required=True)
    prepare_parser.add_argument("--mapping-output", type=Path, required=True)
    score_parser = sub.add_parser("score")
    score_parser.add_argument("--plan", type=Path, required=True)
    score_parser.add_argument("--mapping", type=Path, required=True)
    score_parser.add_argument("--judgments", type=Path, required=True)
    score_parser.add_argument("--output", type=Path, required=True)
    args = parser.parse_args()
    try:
        if args.command == "prepare":
            blind, mapping = prepare(load_json(args.plan), load_json(args.runs))
            write_json(args.blind_output, blind)
            write_json(args.mapping_output, mapping)
            result = {"status": "prepared", "candidate_count": blind["candidate_count"]}
        else:
            result = score(
                load_json(args.plan),
                load_json(args.mapping),
                load_json(args.judgments),
            )
            write_json(args.output, result)
        print(json.dumps(result, indent=2))
        return 0
    except EvaluationError as exc:
        print(json.dumps({"status": "error", "error": str(exc)}, indent=2))
        return 1


if __name__ == "__main__":
    raise SystemExit(main())

SHA-256: 973f6356a5c2db068018d1d1273b1767dcd8dc04be9027d295a63153e6ca2178