← Files Empire LLM for CodexARCHIVED FILE

scripts/empire_benchmarks.py

32.4 KB · Oct 2, 2026 · 00:29 UTC

↓ Download file

#!/usr/bin/env python3
"""Rank current language models and render a sandboxed Codex chart fragment."""

from __future__ import annotations

import argparse
import base64
import hashlib
import html
import json
import math
import sys
from datetime import datetime, timezone
from pathlib import Path
from typing import Any

import empire_review_runtime as runtime


class BenchmarkError(RuntimeError):
    """Raised when benchmark evidence cannot be normalized safely."""


PROFILE_WEIGHTS: dict[str, dict[str, float]] = {
    "agentic": {
        "agentic": 0.42,
        "coding": 0.22,
        "intelligence": 0.16,
        "modality": 0.10,
        "speed": 0.05,
        "cost": 0.05,
    },
    "coding": {
        "agentic": 0.22,
        "coding": 0.42,
        "intelligence": 0.16,
        "modality": 0.10,
        "speed": 0.05,
        "cost": 0.05,
    },
    "balanced": {
        "agentic": 0.25,
        "coding": 0.25,
        "intelligence": 0.20,
        "modality": 0.10,
        "speed": 0.10,
        "cost": 0.10,
    },
    "value": {
        "agentic": 0.18,
        "coding": 0.18,
        "intelligence": 0.14,
        "modality": 0.10,
        "speed": 0.15,
        "cost": 0.25,
    },
}
DIMENSIONS = tuple(next(iter(PROFILE_WEIGHTS.values())))
MAX_TOP = 20
MAX_FRAGMENT_BYTES = 1_000_000


def finite_number(value: Any) -> float | None:
    if isinstance(value, bool):
        return None
    try:
        result = float(value)
    except (TypeError, ValueError):
        return None
    return result if math.isfinite(result) else None


def nested_number(value: Any, *paths: str) -> float | None:
    for path in paths:
        current = value
        for part in path.split("."):
            if not isinstance(current, dict):
                current = None
                break
            current = current.get(part)
        result = finite_number(current)
        if result is not None:
            return result
    return None


def load_object(path: Path) -> dict[str, Any]:
    try:
        value = json.loads(path.read_text(encoding="utf-8"))
    except (OSError, json.JSONDecodeError) as exc:
        raise BenchmarkError(f"Could not read JSON input: {path}") from exc
    if not isinstance(value, dict):
        raise BenchmarkError(f"JSON input must contain an object: {path}")
    return value


def parse_weights(profile: str, overrides: list[str]) -> dict[str, float]:
    weights = dict(PROFILE_WEIGHTS[profile])
    for override in overrides:
        name, separator, raw_value = override.partition("=")
        if not separator or name not in DIMENSIONS:
            raise BenchmarkError(
                "Weights must use agentic|coding|intelligence|modality|speed|cost=VALUE"
            )
        value = finite_number(raw_value)
        if value is None or value < 0 or value > 100:
            raise BenchmarkError(f"Invalid weight for {name}: {raw_value}")
        weights[name] = value
    total = sum(weights.values())
    if total <= 0:
        raise BenchmarkError("At least one scoring weight must be positive")
    return {name: round(value / total, 8) for name, value in weights.items()}


def normalized_modalities(value: Any, direction: str) -> list[str]:
    if isinstance(value, dict):
        candidates = (
            value.get(direction)
            or value.get(f"{direction}_modalities")
            or value.get(f"{direction}s")
            or []
        )
    elif isinstance(value, list):
        candidates = value
    else:
        candidates = []
    if isinstance(candidates, str):
        candidates = [candidates]
    if not isinstance(candidates, list):
        return []
    result: list[str] = []
    for item in candidates:
        if not isinstance(item, str):
            continue
        normalized = item.strip().lower().replace("_", "-")
        if normalized and normalized not in result:
            result.append(normalized)
    return result


def openrouter_index(payload: dict[str, Any]) -> dict[str, dict[str, Any]]:
    return runtime.exact_model_index(payload, "id")


def modality_evidence(
    aa_row: dict[str, Any],
    route: dict[str, Any] | None,
    required_input: list[str],
    required_output: list[str],
) -> dict[str, Any]:
    if route is not None:
        architecture = route.get("architecture")
        architecture = architecture if isinstance(architecture, dict) else {}
        raw_inputs = architecture.get("input_modalities")
        raw_outputs = architecture.get("output_modalities")
        inputs = normalized_modalities(raw_inputs, "input") if isinstance(raw_inputs, list) and all(isinstance(v, str) for v in raw_inputs) else []
        outputs = normalized_modalities(raw_outputs, "output") if isinstance(raw_outputs, list) and all(isinstance(v, str) for v in raw_outputs) else []
        source = "openrouter_exact_model"
    else:
        modalities = aa_row.get("modalities")
        inputs = normalized_modalities(modalities, "input")
        outputs = normalized_modalities(modalities, "output")
        source = "artificial_analysis"
    missing_input = sorted(set(required_input) - set(inputs))
    missing_output = sorted(set(required_output) - set(outputs))
    verified = bool(inputs and outputs)
    compatible = verified and not missing_input and not missing_output
    return {
        "status": "compatible"
        if compatible
        else "incompatible"
        if verified
        else "unverified",
        "source": source,
        "input_modalities": inputs,
        "output_modalities": outputs,
        "missing_input": missing_input,
        "missing_output": missing_output,
    }


def _scale(values: list[float | None], *, reverse: bool = False) -> list[float | None]:
    present = [value for value in values if value is not None]
    if not present:
        return [None for _ in values]
    low, high = min(present), max(present)
    scaled: list[float | None] = []
    for value in values:
        if value is None:
            scaled.append(None)
            continue
        if high == low:
            scaled.append(1.0)
        else:
            score = (value - low) / (high - low)
            scaled.append(1.0 - score if reverse else score)
    return scaled


def _index_score(value: float | None) -> float | None:
    if value is None:
        return None
    return max(0.0, min(1.0, value / 100.0))


def _model_icon(model_id: str, name: str) -> dict[str, Any]:
    icon = runtime.resolve_model_icon(model_id, name)
    return {
        "id": icon.get("icon_id"),
        "svg_path": icon.get("icon_path"),
        "footnote_path": icon.get("icon_footnote_path"),
        "fallback": bool(icon.get("icon_fallback_used")),
    }


def rank_models(
    artificial_analysis: dict[str, Any],
    openrouter: dict[str, Any] | None,
    *,
    profile: str,
    weights: dict[str, float],
    required_input: list[str],
    required_output: list[str],
    include_openai: bool,
    route_qualified_only: bool,
    top: int,
) -> dict[str, Any]:
    if not isinstance(artificial_analysis.get("data"), list):
        raise BenchmarkError("Artificial Analysis payload has no model data")
    route_identity_available = any(
        isinstance(row, dict)
        and isinstance(row.get("openrouter_api_id"), str)
        and bool(row["openrouter_api_id"].strip())
        for row in artificial_analysis["data"]
    )
    if route_qualified_only and artificial_analysis["data"] and not route_identity_available:
        raise BenchmarkError(
            "Exact route qualification requires openrouter_api_id evidence. "
            "Artificial Analysis Free omits this field; use Pro/Commercial access "
            "or omit --route-qualified-only for an unverified research view. "
            "No route has been authorized."
        )
    routes = openrouter_index(openrouter or {})
    aa_routes = runtime.aa_index(artificial_analysis)
    candidates: list[dict[str, Any]] = []
    for row in artificial_analysis["data"]:
        if not isinstance(row, dict):
            continue
        model_id = row.get("id")
        name = row.get("name")
        slug = row.get("slug")
        creator = row.get("model_creator")
        creator = creator if isinstance(creator, dict) else {}
        creator_name = str(creator.get("name") or "Unknown")
        if (
            not isinstance(model_id, str)
            or not model_id.strip()
            or not isinstance(name, str)
            or not name.strip()
            or not isinstance(slug, str)
            or not slug.strip()
        ):
            continue
        is_openai = creator_name.strip().lower() == "openai"
        if is_openai and not include_openai:
            continue
        evaluations = row.get("evaluations")
        evaluations = evaluations if isinstance(evaluations, dict) else {}
        route_id = row.get("openrouter_api_id")
        route_id = route_id if isinstance(route_id, str) else ""
        route = routes.get(route_id) if route_id in aa_routes else None
        modality = modality_evidence(
            row, route, required_input, required_output
        )
        exact_join = route is not None
        route_eligible = bool(
            exact_join and modality["status"] == "compatible" and not is_openai
        )
        if route_qualified_only and not route_eligible:
            continue
        pricing = row.get("pricing")
        pricing = pricing if isinstance(pricing, dict) else {}
        performance = row.get("performance")
        performance = performance if isinstance(performance, dict) else {}
        cost_field = "artificial_analysis_intelligence_index_cost.cost_per_task.total_cost"
        cost = nested_number(row, cost_field)
        if cost is not None and cost < 0:
            cost = None
        speed = nested_number(
            performance,
            "median_output_tokens_per_second",
        )
        if speed is None:
            speed = nested_number(row, "median_output_tokens_per_second")
        candidates.append(
            {
                "aa_id": model_id,
                "name": name,
                "slug": slug,
                "creator": creator_name,
                "release_date": row.get("release_date"),
                "reasoning_model": row.get("reasoning_model"),
                "openrouter_model_id": route_id or None,
                "openrouter_exact_match": exact_join,
                "route_eligible": route_eligible,
                "route_eligibility_reason": "exact_identity_and_modalities"
                if route_eligible
                else "openai_is_codex_lead"
                if is_openai
                else "modality_requirement_failed"
                if modality["status"] == "incompatible"
                else "modality_evidence_unverified"
                if exact_join and modality["status"] == "unverified"
                else "exact_openrouter_identity_unavailable",
                "modality": modality,
                "cost_evidence": {
                    "unit": "USD/task", "source_field": cost_field,
                    "status": "available" if cost is not None else "unavailable",
                },
                "metrics": {
                    "agentic_index": nested_number(
                        evaluations, "artificial_analysis_agentic_index"
                    ),
                    "coding_index": nested_number(
                        evaluations, "artificial_analysis_coding_index"
                    ),
                    "intelligence_index": nested_number(
                        evaluations, "artificial_analysis_intelligence_index"
                    ),
                    "output_tokens_per_second": speed,
                    "cost_per_task_usd": cost,
                    "input_price_per_million_usd": nested_number(
                        pricing, "price_1m_input_tokens"
                    ),
                    "output_price_per_million_usd": nested_number(
                        pricing, "price_1m_output_tokens"
                    ),
                },
                "icon": _model_icon(route_id or f"{creator_name}/{slug}", name),
            }
        )
    if not candidates:
        raise BenchmarkError("No models satisfy the requested benchmark filters")

    speed_scores = _scale(
        [item["metrics"]["output_tokens_per_second"] for item in candidates]
    )
    cost_scores = _scale(
        [item["metrics"]["cost_per_task_usd"] for item in candidates],
        reverse=True,
    )
    for item, speed_score, cost_score in zip(
        candidates, speed_scores, cost_scores, strict=True
    ):
        metrics = item["metrics"]
        dimensions = {
            "agentic": _index_score(metrics["agentic_index"]),
            "coding": _index_score(metrics["coding_index"]),
            "intelligence": _index_score(metrics["intelligence_index"]),
            "modality": 1.0
            if item["modality"]["status"] == "compatible"
            else 0.0
            if item["modality"]["status"] == "incompatible"
            else None,
            "speed": speed_score,
            "cost": cost_score,
        }
        weighted = sum(
            weights[name] * value
            for name, value in dimensions.items()
            if value is not None
        )
        coverage = sum(
            weights[name] for name, value in dimensions.items() if value is not None
        )
        item["dimension_scores"] = {
            name: round(value, 6) if value is not None else None
            for name, value in dimensions.items()
        }
        item["evidence_coverage"] = round(coverage, 6)
        item["weighted_score"] = round(weighted * 100, 2)

    candidates.sort(
        key=lambda item: (
            -item["weighted_score"],
            -item["evidence_coverage"],
            item["name"].lower(),
        )
    )
    for index, item in enumerate(candidates, start=1):
        item["rank"] = index
    visible = candidates[:top]
    aa_source = artificial_analysis.get("_empire_source")
    aa_source = aa_source if isinstance(aa_source, dict) else {}
    return {
        "schema_version": "1.0.0",
        "status": "ranked",
        "generated_at": datetime.now(timezone.utc).isoformat(),
        "profile": profile,
        "weights": weights,
        "requirements": {
            "input_modalities": required_input,
            "output_modalities": required_output,
            "route_qualified_only": route_qualified_only,
            "openai_partner_candidates": False,
        },
        "source_receipts": {
            "artificial_analysis": {
                "status": "available",
                "endpoint": aa_source.get("endpoint"),
                "tier": artificial_analysis.get("tier"),
                "intelligence_index_version": artificial_analysis.get(
                    "intelligence_index_version"
                ),
                "pages_fetched": aa_source.get("pages_fetched"),
                "model_count": len(artificial_analysis["data"]),
                "route_identity_evidence": "available" if route_identity_available else "unavailable",
                "fetched_at": aa_source.get("fetched_at"),
                "rate_limit": aa_source.get("rate_limit"),
                "attribution_url": "https://artificialanalysis.ai/",
            },
            "openrouter": {
                "status": "available" if openrouter is not None else "unavailable",
                "endpoint": runtime.OPENROUTER_USER_MODELS
                if openrouter is not None
                else None,
                "model_count": len(routes),
                "exact_match_count": sum(
                    item["openrouter_exact_match"] for item in candidates
                ),
            },
        },
        "model_count_considered": len(candidates),
        "models": visible,
        "notes": [
            "Weighted scores are Empire routing inferences, not Artificial Analysis rankings.",
            "Missing evidence reduces the score instead of being treated as zero.",
            "Cost scores compare measured USD/task only; token prices are never substituted.",
            "Only exact Artificial Analysis openrouter_api_id joins can become route eligible.",
            "OpenAI models may be shown as research baselines but are never Empire partner candidates.",
        ],
    }


def fetch_sources(args: argparse.Namespace) -> tuple[dict[str, Any], dict[str, Any] | None]:
    if args.aa_fixture:
        aa = load_object(args.aa_fixture.resolve())
        aa.setdefault(
            "_empire_source",
            {
                "endpoint": "offline_fixture",
                "pages_fetched": 1,
                "model_count": len(aa.get("data", [])),
                "fetched_at": None,
                "rate_limit": None,
            },
        )
    else:
        aa_key, aa_source = runtime.resolve_credential(
            ("ARTIFICIAL_ANALYSIS_API_KEY", "AA_API_KEY"), "artificial-analysis"
        )
        if not aa_key:
            raise BenchmarkError(
                runtime.credential_required_message(
                    "Artificial Analysis",
                    aa_source,
                    "run the Empire settings setup before requesting live benchmarks",
                )
            )
        aa = runtime.fetch_artificial_analysis_models(aa_key, access=args.aa_access)

    if args.no_openrouter:
        return aa, None
    if args.openrouter_fixture:
        return aa, load_object(args.openrouter_fixture.resolve())
    openrouter_key, openrouter_source = runtime.resolve_credential(
        ("OPENROUTER_API_KEY",), "openrouter"
    )
    if not openrouter_key:
        if args.route_qualified_only:
            raise BenchmarkError(
                runtime.credential_required_message(
                    "OpenRouter",
                    openrouter_source,
                    "run the Empire settings setup for route-qualified rankings",
                )
            )
        return aa, None
    return aa, runtime.request_json(
        runtime.OPENROUTER_USER_MODELS, api_key=openrouter_key, timeout=10
    )


def markdown_view(payload: dict[str, Any]) -> str:
    source = payload["source_receipts"]["artificial_analysis"]
    lines = [
        f"## Empire benchmark view · {payload['profile']}",
        "",
        (
            "Live Artificial Analysis data"
            if source.get("endpoint") != "offline_fixture"
            else "Offline fixture data"
        )
        + f" · index v{source.get('intelligence_index_version') or 'unknown'}",
        "",
        "| Rank | Model | Weighted | Agentic | Coding | Intelligence | OpenRouter |",
        "| ---: | --- | ---: | ---: | ---: | ---: | --- |",
    ]
    for item in payload["models"]:
        metrics = item["metrics"]
        route = item.get("openrouter_model_id") or "unverified"
        values = [
            item["rank"],
            item["name"],
            f"{item['weighted_score']:.2f}",
            metrics.get("agentic_index"),
            metrics.get("coding_index"),
            metrics.get("intelligence_index"),
            route,
        ]
        cells = [html.escape(str(value if value is not None else "—")) for value in values]
        lines.append("| " + " | ".join(cells) + " |")
    lines.extend(
        (
            "",
            "Weighted scores are Empire inferences. Benchmark source: "
            "[Artificial Analysis](https://artificialanalysis.ai/).",
        )
    )
    return "\n".join(lines) + "\n"


def icon_data_uri(path_value: Any) -> str | None:
    if not isinstance(path_value, str):
        return None
    path = Path(path_value)
    if not path.is_file() or path.is_symlink():
        return None
    data = path.read_bytes()
    if len(data) > 250_000:
        return None
    mime = "image/png" if path.suffix.lower() == ".png" else "image/svg+xml"
    return f"data:{mime};base64,{base64.b64encode(data).decode('ascii')}"


def render_fragment(payload: dict[str, Any]) -> str:
    models = payload.get("models")
    weights = payload.get("weights")
    if not isinstance(models, list) or not models:
        raise BenchmarkError("Benchmark result has no models to render")
    if not isinstance(weights, dict) or set(weights) != set(DIMENSIONS):
        raise BenchmarkError("Benchmark result has invalid weights")
    chart_models: list[dict[str, Any]] = []
    for item in models:
        if not isinstance(item, dict) or not isinstance(item.get("name"), str):
            raise BenchmarkError("Benchmark result contains an invalid model")
        chart_models.append(
            {
                "name": item["name"][:120],
                "creator": str(item.get("creator") or "Unknown")[:80],
                "modelId": str(item.get("openrouter_model_id") or item.get("slug"))[:200],
                "dimensions": item.get("dimension_scores"),
                "metrics": item.get("metrics"),
                "routeEligible": bool(item.get("route_eligible")),
                "modality": item.get("modality", {}).get("status"),
                "icon": icon_data_uri(item.get("icon", {}).get("footnote_path")),
            }
        )
    identity = hashlib.sha256(
        json.dumps(chart_models, sort_keys=True).encode("utf-8")
    ).hexdigest()[:12]
    root_id = f"empire-benchmarks-{identity}"
    data_json = json.dumps(
        {"models": chart_models, "weights": weights},
        separators=(",", ":"),
        ensure_ascii=True,
    ).replace("<", "\\u003c")
    profile = html.escape(str(payload.get("profile") or "weighted").title())
    source = payload.get("source_receipts", {}).get("artificial_analysis", {})
    version = html.escape(str(source.get("intelligence_index_version") or "unknown"))
    endpoint = source.get("endpoint")
    if endpoint == "offline_fixture":
        source_label = "Synthetic fixture"
    else:
        tier = str(source.get("tier") or "unknown").strip().title()
        source_label = f"Artificial Analysis {tier}"
    fetched_at = source.get("fetched_at")
    retrieved_label = (
        f" · retrieved {html.escape(str(fetched_at)[:40])}"
        if isinstance(fetched_at, str) and fetched_at.strip()
        else ""
    )
    fragment = f'''<section id="{root_id}">
  <h1>{profile} model benchmark</h1>
  <p class="text-muted">{html.escape(source_label)} · index v{version}{retrieved_label} · Empire weighted score (0–100)</p>
  <div class="viz-controls" aria-label="Scoring weights">
    <label class="form-label">Agentic <output data-value="agentic"></output><input class="form-range" data-weight="agentic" type="range" min="0" max="100" step="1"></label>
    <label class="form-label">Coding <output data-value="coding"></output><input class="form-range" data-weight="coding" type="range" min="0" max="100" step="1"></label>
    <label class="form-label">Intelligence <output data-value="intelligence"></output><input class="form-range" data-weight="intelligence" type="range" min="0" max="100" step="1"></label>
    <label class="form-label">Modality <output data-value="modality"></output><input class="form-range" data-weight="modality" type="range" min="0" max="100" step="1"></label>
    <label class="form-label">Speed <output data-value="speed"></output><input class="form-range" data-weight="speed" type="range" min="0" max="100" step="1"></label>
    <label class="form-label">Cost <output data-value="cost"></output><input class="form-range" data-weight="cost" type="range" min="0" max="100" step="1"></label>
  </div>
  <div class="empire-aa-axis text-small" aria-hidden="true"><span>100</span><span>75</span><span>50</span><span>25</span><span>0</span></div>
  <div class="empire-aa-bars" role="img" aria-label="Vertical bar chart ranking models by the selected weighted score"></div>
  <p class="empire-aa-detail text-small" aria-live="polite"></p>
  <div class="viz-row"><button class="btn" type="button" data-compare>Ask Codex to compare the leaders</button></div>
  <p class="text-small text-muted">Benchmark data © Artificial Analysis. Weighted scores are Empire inferences; exact OpenRouter identity and modality checks determine route eligibility.</p>
  <script type="application/json" data-benchmark-data>{data_json}</script>
  <style>
    #{root_id} {{ position: relative; color: var(--foreground); }}
    #{root_id} .empire-aa-axis {{ position: absolute; inset: 15.1rem auto 5.4rem 0; display: flex; flex-direction: column; justify-content: space-between; color: var(--muted-foreground); }}
    #{root_id} .empire-aa-bars {{ min-height: 22rem; margin-left: 2.25rem; display: grid; grid-template-columns: repeat(auto-fit, minmax(5.5rem, 1fr)); gap: 0.75rem; align-items: end; border-bottom: 1px solid var(--border); }}
    #{root_id} .empire-aa-model {{ min-width: 0; display: grid; grid-template-rows: 15rem auto auto; gap: 0.45rem; align-items: end; text-align: center; }}
    #{root_id} .empire-aa-bar-zone {{ height: 15rem; display: flex; align-items: end; justify-content: center; background: repeating-linear-gradient(to top, transparent 0, transparent calc(25% - 1px), var(--border) 25%); }}
    #{root_id} .empire-aa-bar {{ width: min(3.25rem, 70%); min-height: 2px; background: var(--viz-series-1); transition: height 180ms ease; }}
    #{root_id} .empire-aa-score {{ color: var(--foreground); }}
    #{root_id} .empire-aa-chip {{ justify-self: center; max-width: 100%; overflow-wrap: anywhere; }}
    #{root_id} .empire-aa-chip img {{ width: 14px; height: 14px; flex: none; }}
    #{root_id} .empire-aa-detail {{ min-height: 1.4em; color: var(--muted-foreground); }}
    @media (prefers-reduced-motion: reduce) {{ #{root_id} .empire-aa-bar {{ transition: none; }} }}
    @media (max-width: 420px) {{ #{root_id} .empire-aa-bars {{ grid-template-columns: repeat(2, minmax(0, 1fr)); }} #{root_id} .empire-aa-axis {{ display: none; }} }}
  </style>
  <script>
    (() => {{
      const root = document.getElementById('{root_id}');
      const data = JSON.parse(root.querySelector('[data-benchmark-data]').textContent);
      const bars = root.querySelector('.empire-aa-bars');
      const detail = root.querySelector('.empire-aa-detail');
      const sliders = [...root.querySelectorAll('[data-weight]')];
      sliders.forEach((input) => {{ input.value = Math.round(data.weights[input.dataset.weight] * 100); }});
      const currentWeights = () => {{
        const raw = Object.fromEntries(sliders.map((input) => [input.dataset.weight, Number(input.value)]));
        const total = Object.values(raw).reduce((sum, value) => sum + value, 0) || 1;
        return Object.fromEntries(Object.entries(raw).map(([key, value]) => [key, value / total]));
      }};
      const score = (model, weights) => Object.entries(weights).reduce((sum, [key, weight]) => {{
        const value = model.dimensions && model.dimensions[key];
        return sum + (typeof value === 'number' ? value * weight : 0);
      }}, 0) * 100;
      const render = () => {{
        const weights = currentWeights();
        sliders.forEach((input) => {{ root.querySelector(`[data-value="${{input.dataset.weight}}"]`).textContent = `${{Math.round(weights[input.dataset.weight] * 100)}}%`; }});
        const ranked = data.models.map((model) => ({{...model, score: score(model, weights)}})).sort((a, b) => b.score - a.score || a.name.localeCompare(b.name));
        bars.replaceChildren();
        ranked.forEach((model, index) => {{
          const item = document.createElement('article'); item.className = 'empire-aa-model';
          const zone = document.createElement('div'); zone.className = 'empire-aa-bar-zone';
          const bar = document.createElement('div'); bar.className = 'empire-aa-bar'; bar.style.height = `${{Math.max(0, Math.min(100, model.score))}}%`; bar.setAttribute('data-tooltip', `${{model.name}} · ${{model.score.toFixed(2)}} weighted score`); zone.append(bar);
          const value = document.createElement('strong'); value.className = 'empire-aa-score tabular-nums'; value.textContent = model.score.toFixed(1);
          const chip = document.createElement('div'); chip.className = 'viz-badge empire-aa-chip';
          if (model.icon) {{ const image = document.createElement('img'); image.src = model.icon; image.alt = ''; image.setAttribute('aria-hidden', 'true'); chip.append(image); }}
          const label = document.createElement('span'); label.textContent = model.name; chip.append(label);
          item.append(zone, value, chip); bars.append(item);
          if (index === 0) {{ detail.textContent = `Leader: ${{model.name}} · ${{model.score.toFixed(2)}} · ${{model.routeEligible ? 'exact OpenRouter route eligible' : 'benchmark view only'}} · modality ${{model.modality || 'unverified'}}`; }}
        }});
      }};
      sliders.forEach((input) => input.addEventListener('input', render));
      root.querySelector('[data-compare]').addEventListener('click', async () => {{
        const weights = currentWeights();
        const ranked = data.models.map((model) => ({{...model, score: score(model, weights)}})).sort((a, b) => b.score - a.score);
        const leaders = ranked.slice(0, 2).map((model) => `${{model.name}} (${{model.modelId}})`).join(' and ');
        if (window.openai && window.openai.sendFollowUpMessage) {{
          await window.openai.sendFollowUpMessage({{title: 'Compare benchmark leaders', prompt: `Use Empire benchmark evidence to compare ${{leaders}} for my project. Recheck current Artificial Analysis and OpenRouter data before recommending a route; do not dispatch a paid completion.`}});
        }} else {{ detail.textContent = `Compare ${{leaders}} in a Codex task; this surface cannot send a follow-up here.`; }}
      }});
      render();
    }})();
  </script>
</section>
'''
    if len(fragment.encode("utf-8")) > MAX_FRAGMENT_BYTES:
        raise BenchmarkError("Rendered benchmark fragment exceeds the 1 MB ceiling")
    if any(marker in fragment for marker in ("fetch(", "XMLHttpRequest", "WebSocket")):
        raise BenchmarkError("Rendered benchmark fragment attempted network access")
    return fragment


def add_source_arguments(parser: argparse.ArgumentParser) -> None:
    parser.add_argument("--aa-access", choices=("auto", "free", "pro"), default="auto")
    parser.add_argument("--aa-fixture", type=Path)
    parser.add_argument("--openrouter-fixture", type=Path)
    parser.add_argument("--no-openrouter", action="store_true")
    parser.add_argument("--profile", choices=tuple(PROFILE_WEIGHTS), default="agentic")
    parser.add_argument("--weight", action="append", default=[])
    parser.add_argument("--input-modality", action="append", default=[])
    parser.add_argument("--output-modality", action="append", default=[])
    parser.add_argument("--include-openai", action="store_true")
    parser.add_argument("--route-qualified-only", action="store_true")
    parser.add_argument("--top", type=int, default=8)


def parser() -> argparse.ArgumentParser:
    root = argparse.ArgumentParser(description=__doc__)
    commands = root.add_subparsers(dest="command", required=True)
    rank = commands.add_parser("rank", help="Fetch and rank current models")
    add_source_arguments(rank)
    rank.add_argument("--view", choices=("json", "markdown"), default="json")
    rank.add_argument("--output", type=Path)
    render = commands.add_parser("render", help="Render ranked JSON as a Codex HTML fragment")
    render.add_argument("--input", type=Path, required=True)
    render.add_argument("--output", type=Path, required=True)
    return root


def main(argv: list[str] | None = None) -> int:
    args = parser().parse_args(argv)
    try:
        if args.command == "rank":
            if args.top < 1 or args.top > MAX_TOP:
                raise BenchmarkError(f"--top must be between 1 and {MAX_TOP}")
            weights = parse_weights(args.profile, args.weight)
            aa, openrouter = fetch_sources(args)
            result = rank_models(
                aa,
                openrouter,
                profile=args.profile,
                weights=weights,
                required_input=args.input_modality or ["text"],
                required_output=args.output_modality or ["text"],
                include_openai=args.include_openai,
                route_qualified_only=args.route_qualified_only,
                top=args.top,
            )
            output = (
                markdown_view(result)
                if args.view == "markdown"
                else json.dumps(result, indent=2) + "\n"
            )
            if args.output:
                args.output.resolve().write_text(output, encoding="utf-8")
            else:
                sys.stdout.write(output)
            return 0
        payload = load_object(args.input.resolve())
        fragment = render_fragment(payload)
        args.output.resolve().write_text(fragment, encoding="utf-8")
        print(json.dumps({"status": "rendered", "path": str(args.output.resolve())}))
        return 0
    except (BenchmarkError, runtime.RouterError) as exc:
        print(json.dumps({"status": "error", "error": str(exc)}), file=sys.stderr)
        return 2


if __name__ == "__main__":
    raise SystemExit(main())

SHA-256: c62ed9a7995b86d92f45b24817f429b193f5c6f178d36bd456340e36d779be9d