← Files Model CompassARCHIVED FILE

skills/compare-model-tradeoffs/references/openai-baseline.json

5.72 KB · Oct 4, 2026 · 12:30 UTC

↓ Download file

{
  "schemaVersion": 1,
  "asOf": "2026-07-24",
  "comparisonBaseline": {
    "preset": "Power",
    "model": "gpt-5.6-sol",
    "reasoningEffort": "medium",
    "serviceTier": "standard",
    "source": "codex-models"
  },
  "sources": {
    "codex-models": "https://learn.chatgpt.com/docs/models",
    "codex-speed": "https://learn.chatgpt.com/docs/agent-configuration/speed",
    "codex-pricing": "https://learn.chatgpt.com/docs/pricing",
    "api-pricing": "https://developers.openai.com/api/docs/pricing",
    "gpt-5.6-launch": "https://openai.com/index/gpt-5-6/",
    "genebench-pro": "https://cdn.openai.com/pdf/21938268-21af-442f-af93-3b2249afb241/genebench-pro.pdf",
    "app-server-model-list": "https://learn.chatgpt.com/docs/app-server#list-models-modellist"
  },
  "fastMode": {
    "modelSpeedMultiplier": 1.5,
    "chatgptCreditMultipliers": {
      "gpt-5.6": 2.5,
      "gpt-5.5": 2.5,
      "gpt-5.4": 2
    },
    "apiPriorityPriceMultiplierForGpt56": 2,
    "scopeNote": "The 1.5x claim applies to model speed, not guaranteed end-to-end task time."
  },
  "models": {
    "gpt-5.6-sol": {
      "capabilityOrdinal": 5,
      "speedOrdinal": 2,
      "apiUsdPerMillion": {
        "input": 5,
        "cachedInput": 0.5,
        "cacheWrite": 6.25,
        "output": 30
      },
      "codexCreditsPerMillion": {
        "input": 125,
        "cachedInput": 12.5,
        "output": 750
      },
      "publishedBenchmarks": {
        "aaCodingAgentIndexV1_1": 80,
        "deepSWEV1_1": 72.7,
        "terminalBenchV2_1": 88.8,
        "agentsLastExam": 52.7
      },
      "geneBenchPro": {
        "none": { "passPercent": 3.7, "averageTokensThousands": 1.4 },
        "low": { "passPercent": 14.4, "averageTokensThousands": 5.6 },
        "medium": { "passPercent": 22.5, "averageTokensThousands": 14.4 },
        "high": { "passPercent": 24.4, "averageTokensThousands": 19.5 },
        "xhigh": { "passPercent": 26.8, "averageTokensThousands": 25.7 },
        "max": { "passPercent": 28.7, "averageTokensThousands": 33.2 }
      }
    },
    "gpt-5.6-terra": {
      "capabilityOrdinal": 4,
      "speedOrdinal": 3,
      "apiUsdPerMillion": {
        "input": 2.5,
        "cachedInput": 0.25,
        "cacheWrite": 3.125,
        "output": 15
      },
      "codexCreditsPerMillion": {
        "input": 62.5,
        "cachedInput": 6.25,
        "output": 375
      },
      "publishedBenchmarks": {
        "aaCodingAgentIndexV1_1": 77.4,
        "deepSWEV1_1": 69.6,
        "terminalBenchV2_1": 87.4,
        "agentsLastExam": 50.4
      },
      "geneBenchPro": {
        "none": { "passPercent": 1, "averageTokensThousands": 0.93 },
        "low": { "passPercent": 6.5, "averageTokensThousands": 5.5 },
        "medium": { "passPercent": 13.6, "averageTokensThousands": 15.9 },
        "high": { "passPercent": 16.2, "averageTokensThousands": 22.2 },
        "xhigh": { "passPercent": 18.8, "averageTokensThousands": 31.1 },
        "max": { "passPercent": 23.3, "averageTokensThousands": 54.3 }
      }
    },
    "gpt-5.6-luna": {
      "capabilityOrdinal": 3,
      "speedOrdinal": 4,
      "apiUsdPerMillion": {
        "input": 1,
        "cachedInput": 0.1,
        "cacheWrite": 1.25,
        "output": 6
      },
      "codexCreditsPerMillion": {
        "input": 25,
        "cachedInput": 2.5,
        "output": 150
      },
      "publishedBenchmarks": {
        "aaCodingAgentIndexV1_1": 74.6,
        "deepSWEV1_1": 67.2,
        "terminalBenchV2_1": 84.7,
        "agentsLastExam": 50.3
      },
      "geneBenchPro": {
        "none": { "passPercent": 0.8, "averageTokensThousands": 0.975 },
        "low": { "passPercent": 2.3, "averageTokensThousands": 3.6 },
        "medium": { "passPercent": 4.7, "averageTokensThousands": 15.6 },
        "high": { "passPercent": 8, "averageTokensThousands": 32.3 },
        "xhigh": { "passPercent": 10.8, "averageTokensThousands": 53.1 },
        "max": { "passPercent": 16.5, "averageTokensThousands": 118.2 }
      }
    },
    "gpt-5.5": {
      "capabilityOrdinal": 4,
      "speedOrdinal": 3,
      "apiUsdPerMillion": {
        "input": 5,
        "cachedInput": 0.5,
        "cacheWrite": null,
        "output": 30
      },
      "codexCreditsPerMillion": {
        "input": 125,
        "cachedInput": 12.5,
        "output": 750
      },
      "publishedBenchmarks": {
        "aaCodingAgentIndexV1_1": 76.4,
        "deepSWEV1_1": 67,
        "terminalBenchV2_1": 85.6,
        "agentsLastExam": 46.9
      }
    },
    "gpt-5.4": {
      "capabilityOrdinal": 3,
      "speedOrdinal": 3,
      "apiUsdPerMillion": {
        "input": 2.5,
        "cachedInput": 0.25,
        "cacheWrite": null,
        "output": 15
      },
      "codexCreditsPerMillion": {
        "input": 62.5,
        "cachedInput": 6.25,
        "output": 375
      }
    },
    "gpt-5.4-mini": {
      "capabilityOrdinal": 2,
      "speedOrdinal": 4,
      "apiUsdPerMillion": {
        "input": 0.75,
        "cachedInput": 0.075,
        "cacheWrite": null,
        "output": 4.5
      },
      "codexCreditsPerMillion": {
        "input": 18.75,
        "cachedInput": 1.875,
        "output": 113
      }
    },
    "gpt-5.3-codex-spark": {
      "capabilityOrdinal": 2,
      "speedOrdinal": 5,
      "apiUsdPerMillion": null,
      "codexCreditsPerMillion": null,
      "note": "Research preview; final rates and comparable public benchmark data are not published."
    }
  },
  "comparability": {
    "publishedBenchmarkEffort": "unknown unless the individual source explicitly labels it",
    "ordinalRatings": "Editorial 1-5 priors, not linear percentages",
    "geneBenchPro": "Domain-specific agentic biology calibration; do not generalize as a coding curve",
    "sweBenchPro": "Omitted because OpenAI later reported substantial task-quality problems in the benchmark"
  }
}

SHA-256: eca27398fd67542340cc506b1817402e5d32538e0475ffd7c15ee92bcf970094