← Files Scientific Visuals & TablesARCHIVED FILE

skills/scientific-visual-table-style/references/openai_visual_reference_library.jsonl

246 KB · Oct 4, 2026 · 12:32 UTC

↓ Download file

{"id": "OAI-VIS-001", "recommended_rank": 1, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Pareto-frontier comparison chart", "title_caption": "Jalapeño widens the lead at previous-best TBT", "what_it_demonstrates": "Canonical sparse frontier chart: one decisive claim, restrained competitors, direct operating-point comparison, and a shareable exact chart anchor.", "source_publication": "Jalapeño: First results", "publication_date": "2026-08-25", "provenance": "OpenAI official research/engineering page", "exact_or_closest_url": "https://openai.com/index/jalapeno-first-results/#chart-31bFDtnrbeUgGQ7fLMZRX4", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/jalapeno-first-results/", "link_precision": "Exact chart anchor", "design_tags": "pareto frontier; latency-throughput; direct comparison; minimal axes", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-002", "recommended_rank": 2, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Score–latency / model comparison", "title_caption": "Agents’ Last Exam", "what_it_demonstrates": "Makes capability and latency jointly legible in a compact frontier-style comparator.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#efficient-by-default-maximum-performance-on-demand", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "score-latency; frontier; multi-model; direct labeling", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-003", "recommended_rank": 3, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Score–cost–time comparison", "title_caption": "Artificial Analysis Intelligence Index v4.1", "what_it_demonstrates": "Shows quality, monetary cost, and elapsed time together without dashboard clutter.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#efficient-by-default-maximum-performance-on-demand", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "cost-performance; time-performance; multi-objective", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-004", "recommended_rank": 4, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Professional benchmark table", "title_caption": "Professional", "what_it_demonstrates": "Canonical launch-table style: grouped rows, compact model columns, clear metric direction, and sparse bolding.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable evaluations section", "design_tags": "benchmark table; no vertical rules; bold best; grouped rows; compact precision", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-005", "recommended_rank": 5, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Diagram", "visual_type": "Context-management diagram", "title_caption": "Append-only context", "what_it_demonstrates": "Near-monochrome structure, one accent, exact alignment, and immediate conceptual legibility.", "source_publication": "GPT‑5.6: Frontier intelligence, more efficiently", "publication_date": "2026-07-09", "provenance": "OpenAI official engineering page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6-frontier-intelligence-efficiency/#avoid-context-bloat", "direct_asset_url": "https://images.ctfassets.net/kftzwdyauwt9/4Yc2D5KqZejneZPTlOtA39/9799e64403c51e214ca4c53d9c8b40ed/Append-only_context_light_desktop.svg?w=3840&q=80", "asset_format": "SVG", "parent_url": "https://openai.com/index/gpt-5-6-frontier-intelligence-efficiency/", "link_precision": "Direct SVG asset + stable section anchor", "design_tags": "context management; SVG; minimal diagram; one accent", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-006", "recommended_rank": 6, "quality_grade": "A+ — canonical", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Performance–cost frontier scatter", "title_caption": "Figure 2: HealthBench score versus inference cost", "what_it_demonstrates": "One of the strongest OpenAI cost-frontier references: log-cost axis, direct model labels, and connected reasoning-effort settings.", "source_publication": "HealthBench: Evaluating Large Language Models Towards Improved Human Health", "publication_date": "2025-05-13", "provenance": "OpenAI-authored benchmark paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2505.08775v1/figures/cost_2.png", "direct_asset_url": "https://arxiv.org/html/2505.08775v1/figures/cost_2.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/healthbench/", "link_precision": "Direct paper image asset", "design_tags": "cost-performance; pareto frontier; log scale; direct labels", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-007", "recommended_rank": 7, "quality_grade": "A+ — canonical", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Reliability line chart", "title_caption": "Figure 7: worst-at-k performance up to k=16", "what_it_demonstrates": "A simple family of monotonic curves that communicates reliability degradation immediately.", "source_publication": "HealthBench: Evaluating Large Language Models Towards Improved Human Health", "publication_date": "2025-05-13", "provenance": "OpenAI-authored benchmark paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2505.08775v1/figures/worst.png", "direct_asset_url": "https://arxiv.org/html/2505.08775v1/figures/worst.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/healthbench/", "link_precision": "Direct paper image asset", "design_tags": "reliability; line chart; worst-of-k; shared scale", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-008", "recommended_rank": 8, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Scatter / trend chart", "title_caption": "Chain-of-thought monitorability versus pretraining scale", "what_it_demonstrates": "A contemporary sparse scatter with fitted trend, model-family labeling, and uncertainty context.", "source_publication": "Evaluating chain-of-thought monitorability", "publication_date": "2025-12-18", "provenance": "OpenAI official safety research page", "exact_or_closest_url": "https://openai.com/index/chain-of-thought-monitorability/#effect-of-pretraining-scale", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/chain-of-thought-monitorability/", "link_precision": "Stable section anchor", "design_tags": "scatter; trend; pretraining scale; monitorability", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-009", "recommended_rank": 9, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost–performance chart", "title_caption": "SEC-Bench Pro", "what_it_demonstrates": "Canonical cyber capability frontier against API cost.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image21.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image21.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "SEC-Bench Pro; cost-performance; cyber", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-010", "recommended_rank": 10, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Heatmap", "title_caption": "Needle-in-a-haystack retrieval across context length and position", "what_it_demonstrates": "One of OpenAI’s strongest matrix visuals: sparse labels, perceptually ordered values, and immediate failure-region visibility.", "source_publication": "Introducing GPT‑4.1 in the API", "publication_date": "2025-04-14", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/gpt-4-1/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-4-1/", "link_precision": "Stable evaluations or research section", "design_tags": "heatmap; long context; matrix; failure regions", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-011", "recommended_rank": 11, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Scaling curves", "title_caption": "o1 performance with train-time and test-time compute", "what_it_demonstrates": "Canonical OpenAI reasoning figure: two simple curves that make the scaling thesis visible in seconds.", "source_publication": "Learning to reason with LLMs", "publication_date": "2024-09-12", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/learning-to-reason-with-llms/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/learning-to-reason-with-llms/", "link_precision": "Stable evaluations or research section", "design_tags": "test-time compute; train-time compute; scaling; two-panel", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-012", "recommended_rank": 12, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Scaling curve", "title_caption": "Test-time compute scaling", "what_it_demonstrates": "A clean monotonic scaling plot with each point representing a full evaluation run.", "source_publication": "BrowseComp: a benchmark for browsing agents", "publication_date": "2025-04-10", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/browsecomp/#test-time-compute-scaling", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/browsecomp/", "link_precision": "Stable section anchor", "design_tags": "test-time compute; scaling curve; agent effort", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-013", "recommended_rank": 13, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Bar chart with uncertainty", "title_caption": "Deployment simulation: rate of undesired behavior", "what_it_demonstrates": "A model-card bar/interval chart with a zero-oriented axis, concise title, and confidence intervals.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image12.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image12.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "deployment simulation; undesired behavior; confidence intervals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-014", "recommended_rank": 14, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Grouped bar chart with intervals", "title_caption": "First-person fairness evaluations", "what_it_demonstrates": "An excellent contemporary grouped bar and confidence-interval reference with careful subgroup structure.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image29.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image29.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "fairness; first-person; confidence intervals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-015", "recommended_rank": 15, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Line / trend chart", "title_caption": "Chain-of-thought controllability versus response length", "what_it_demonstrates": "A clean relationship plot with model separation and minimal annotation.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image42.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image42.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "CoT controllability; response length; trend", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-016", "recommended_rank": 16, "quality_grade": "A+ — canonical", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Grouped bar / line chart", "title_caption": "Figure 4: reasoning effort by price bucket", "what_it_demonstrates": "A strong reference for showing a compute setting across economically meaningful task strata.", "source_publication": "SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?", "publication_date": "2025-02-17", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2502.12115v3/x4.png", "direct_asset_url": "https://arxiv.org/html/2502.12115v3/x4.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/swe-lancer/", "link_precision": "Direct paper image asset", "design_tags": "reasoning effort; price buckets; compute", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-017", "recommended_rank": 17, "quality_grade": "A+ — canonical", "reference_tier": "4 — Historical OpenAI report", "category": "Chart", "visual_type": "Scaling-law plot", "title_caption": "Figure 1: predictable scaling of final loss", "what_it_demonstrates": "A landmark plot tying small-run predictions to the final training run with simple log axes and a clear extrapolation.", "source_publication": "GPT‑4 Technical Report", "publication_date": "2023-03-14", "provenance": "Official OpenAI technical report PDF", "exact_or_closest_url": "https://cdn.openai.com/papers/gpt-4.pdf#page=3", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://cdn.openai.com/papers/gpt-4.pdf", "link_precision": "Exact PDF page (3)", "design_tags": "scaling law; prediction; log axes", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-018", "recommended_rank": 18, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Horizontal stacked bar chart", "title_caption": "Model deliverables judged better, as good as, or worse than expert work", "what_it_demonstrates": "A canonical part-to-whole comparison with neutral outcome categories and direct percentages.", "source_publication": "Measuring the performance of our models on real-world tasks (GDPval)", "publication_date": "2025-09-25", "provenance": "OpenAI official economic research page", "exact_or_closest_url": "https://openai.com/index/gdpval/#early-results", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gdpval/", "link_precision": "Stable section anchor", "design_tags": "stacked bars; human baseline; preference; part-to-whole", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-019", "recommended_rank": 19, "quality_grade": "A+ — canonical", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Calibration plot", "title_caption": "Accuracy versus stated confidence", "what_it_demonstrates": "A strong reliability diagram with an ideal diagonal and concise confidence bins.", "source_publication": "Introducing SimpleQA", "publication_date": "2024-10-30", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/introducing-simpleqa/#results", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-simpleqa/", "link_precision": "Stable results/analysis section", "design_tags": "calibration; reliability; diagonal reference", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-020", "recommended_rank": 20, "quality_grade": "A+ — canonical", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Cost–quality scatter", "title_caption": "Figure 6: JudgeEval F1 versus average cost", "what_it_demonstrates": "A compact cost/quality plot with interpretable operating points for evaluator design.", "source_publication": "PaperBench: Evaluating AI’s Ability to Replicate AI Research", "publication_date": "2025-04-02", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2504.01848#page=17", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/paperbench/", "link_precision": "Exact PDF page", "design_tags": "cost-quality; scatter; evaluator", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-021", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Example table", "title_caption": "BrowseComp example questions", "what_it_demonstrates": "A tabbed example presentation that preserves legibility without exposing a giant table.", "source_publication": "BrowseComp: a benchmark for browsing agents", "publication_date": "2025-04-10", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/browsecomp/#about-the-browsecomp-benchmark", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/browsecomp/", "link_precision": "Stable section anchor", "design_tags": "examples; tabbed display; qualitative table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-022", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Pie chart", "title_caption": "Dataset topic distribution", "what_it_demonstrates": "A rare OpenAI pie chart used only when part-to-whole topic composition is the actual question.", "source_publication": "BrowseComp: a benchmark for browsing agents", "publication_date": "2025-04-10", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/browsecomp/#dataset-diversity-difficulty", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/browsecomp/", "link_precision": "Stable section anchor", "design_tags": "dataset composition; pie chart; limited categories", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-023", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Distribution chart", "title_caption": "Distribution of task pass rates", "what_it_demonstrates": "Shows benchmark heterogeneity and mass at 0% and 100% without over-annotation.", "source_publication": "BrowseComp: a benchmark for browsing agents", "publication_date": "2025-04-10", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/browsecomp/#distribution-of-pass-rates", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/browsecomp/", "link_precision": "Stable section anchor", "design_tags": "distribution; pass rate; difficulty", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-024", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Histogram", "title_caption": "Human search time for solvable and unsolvable problems", "what_it_demonstrates": "Paired distributions that make benchmark difficulty tangible.", "source_publication": "BrowseComp: a benchmark for browsing agents", "publication_date": "2025-04-10", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/browsecomp/#dataset-diversity-difficulty", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/browsecomp/", "link_precision": "Stable section anchor", "design_tags": "histogram; time distribution; human baseline", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-025", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Summary table", "title_caption": "Human verification campaign", "what_it_demonstrates": "A compact denominator-aware table for solvable and unsolvable counts.", "source_publication": "BrowseComp: a benchmark for browsing agents", "publication_date": "2025-04-10", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/browsecomp/#dataset-diversity-difficulty", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/browsecomp/", "link_precision": "Stable section anchor", "design_tags": "summary table; counts; percentages", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-026", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Aggregation comparison chart", "title_caption": "Majority voting, weighted voting, and best-of-N", "what_it_demonstrates": "Compares aggregation methods against a single-attempt baseline with aligned x-values.", "source_publication": "BrowseComp: a benchmark for browsing agents", "publication_date": "2025-04-10", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/browsecomp/#aggregation-strategies-leveraging-additional-compute", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/browsecomp/", "link_precision": "Stable section anchor", "design_tags": "best-of-N; voting; test-time compute", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-027", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Model benchmark table", "title_caption": "Performance of OpenAI models", "what_it_demonstrates": "A tiny, high-signal result table with one metric and a necessary footnote.", "source_publication": "BrowseComp: a benchmark for browsing agents", "publication_date": "2025-04-10", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/browsecomp/#performance-of-openai-models", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/browsecomp/", "link_precision": "Stable section anchor", "design_tags": "benchmark table; one metric; footnote", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-028", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Success-rate comparison chart", "title_caption": "Computer-use success rates", "what_it_demonstrates": "A minimal bar chart for model versus agent or human baselines.", "source_publication": "Computer-Using Agent", "publication_date": "2025-01-23", "provenance": "OpenAI official product/research page", "exact_or_closest_url": "https://openai.com/index/computer-using-agent/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/computer-using-agent/", "link_precision": "Stable evaluations section", "design_tags": "success rate; computer use; bar chart", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-029", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Benchmark table", "title_caption": "OSWorld, WebArena, and WebVoyager", "what_it_demonstrates": "Three related computer-use benchmarks in a compact table with clear baselines.", "source_publication": "Computer-Using Agent", "publication_date": "2025-01-23", "provenance": "OpenAI official product/research page", "exact_or_closest_url": "https://openai.com/index/computer-using-agent/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/computer-using-agent/", "link_precision": "Stable evaluations section", "design_tags": "computer use; benchmark table; baselines", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-030", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Small-multiple scatter", "title_caption": "Monitorability across tasks and model families", "what_it_demonstrates": "Shared axes and repeated facets make a complex evaluation family readable.", "source_publication": "Evaluating chain-of-thought monitorability", "publication_date": "2025-12-18", "provenance": "OpenAI official safety research page", "exact_or_closest_url": "https://openai.com/index/chain-of-thought-monitorability/#results", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/chain-of-thought-monitorability/", "link_precision": "Stable section anchor", "design_tags": "small multiples; scatter; shared scales", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-031", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Paired comparison chart", "title_caption": "Monitoring chain of thought versus actions and outputs", "what_it_demonstrates": "A clean paired comparison that supports a mechanistic claim.", "source_publication": "Evaluating chain-of-thought monitorability", "publication_date": "2025-12-18", "provenance": "OpenAI official safety research page", "exact_or_closest_url": "https://openai.com/index/chain-of-thought-monitorability/#results", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/chain-of-thought-monitorability/", "link_precision": "Stable section anchor", "design_tags": "paired comparison; monitoring; behavior", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-032", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Correlation matrix / table", "title_caption": "Relationships among monitorability metrics", "what_it_demonstrates": "A restrained summary of multiple correlated measures.", "source_publication": "Evaluating chain-of-thought monitorability", "publication_date": "2025-12-18", "provenance": "OpenAI official safety research page", "exact_or_closest_url": "https://openai.com/index/chain-of-thought-monitorability/#results", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/chain-of-thought-monitorability/", "link_precision": "Stable section anchor", "design_tags": "correlation; matrix; metrics", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-033", "recommended_rank": "", "quality_grade": "B — historical", "reference_tier": "4 — Historical OpenAI report", "category": "Chart", "visual_type": "Capability prediction plot", "title_caption": "Figure 2: predicting coding capability", "what_it_demonstrates": "Shows a preregistered capability prediction with uncertainty against the realized point.", "source_publication": "GPT‑4 Technical Report", "publication_date": "2023-03-14", "provenance": "Official OpenAI technical report PDF", "exact_or_closest_url": "https://cdn.openai.com/papers/gpt-4.pdf#page=3", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://cdn.openai.com/papers/gpt-4.pdf", "link_precision": "Exact PDF page (3)", "design_tags": "scaling; coding; prediction interval", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-034", "recommended_rank": "", "quality_grade": "B — historical", "reference_tier": "4 — Historical OpenAI report", "category": "Chart", "visual_type": "Exam percentile comparison", "title_caption": "Figure 3: performance on academic and professional exams", "what_it_demonstrates": "Turns heterogeneous exam scores into an intuitive percentile comparison.", "source_publication": "GPT‑4 Technical Report", "publication_date": "2023-03-14", "provenance": "Official OpenAI technical report PDF", "exact_or_closest_url": "https://cdn.openai.com/papers/gpt-4.pdf#page=6", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://cdn.openai.com/papers/gpt-4.pdf", "link_precision": "Exact PDF page (6)", "design_tags": "exams; percentile; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-035", "recommended_rank": "", "quality_grade": "B — historical", "reference_tier": "4 — Historical OpenAI report", "category": "Chart", "visual_type": "Multilingual comparison chart", "title_caption": "Figure 4: MMLU performance across languages", "what_it_demonstrates": "A compact horizontal comparison across many languages and models.", "source_publication": "GPT‑4 Technical Report", "publication_date": "2023-03-14", "provenance": "Official OpenAI technical report PDF", "exact_or_closest_url": "https://cdn.openai.com/papers/gpt-4.pdf#page=8", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://cdn.openai.com/papers/gpt-4.pdf", "link_precision": "Exact PDF page (8)", "design_tags": "multilingual; horizontal bars; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-036", "recommended_rank": "", "quality_grade": "B — historical", "reference_tier": "4 — Historical OpenAI report", "category": "Chart", "visual_type": "Calibration curve", "title_caption": "Figure 8: calibration on multiple-choice questions", "what_it_demonstrates": "Reliability diagram with an ideal reference line and concise binning.", "source_publication": "GPT‑4 Technical Report", "publication_date": "2023-03-14", "provenance": "Official OpenAI technical report PDF", "exact_or_closest_url": "https://cdn.openai.com/papers/gpt-4.pdf#page=12", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://cdn.openai.com/papers/gpt-4.pdf", "link_precision": "Exact PDF page (12)", "design_tags": "calibration; reliability; diagonal reference", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-037", "recommended_rank": "", "quality_grade": "B — historical", "reference_tier": "4 — Historical OpenAI report", "category": "Table", "visual_type": "Risk evaluation table", "title_caption": "Safety and risk evaluation summary", "what_it_demonstrates": "Dense safety results with explicit baselines, metric direction, and notes.", "source_publication": "GPT‑4 Technical Report", "publication_date": "2023-03-14", "provenance": "Official OpenAI technical report PDF", "exact_or_closest_url": "https://cdn.openai.com/papers/gpt-4.pdf#page=12", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://cdn.openai.com/papers/gpt-4.pdf", "link_precision": "Exact PDF page (12)", "design_tags": "safety; risk; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-038", "recommended_rank": "", "quality_grade": "B — historical", "reference_tier": "4 — Historical OpenAI report", "category": "Table", "visual_type": "Benchmark comparison table", "title_caption": "Table 2: academic benchmark performance", "what_it_demonstrates": "Classic scientific benchmark table with model columns and sparse emphasis.", "source_publication": "GPT‑4 Technical Report", "publication_date": "2023-03-14", "provenance": "Official OpenAI technical report PDF", "exact_or_closest_url": "https://cdn.openai.com/papers/gpt-4.pdf#page=5", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://cdn.openai.com/papers/gpt-4.pdf", "link_precision": "Exact PDF page (5)", "design_tags": "benchmark table; academic; bold best", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-039", "recommended_rank": "", "quality_grade": "B — historical", "reference_tier": "4 — Historical OpenAI report", "category": "Chart", "visual_type": "Safety benchmark comparison", "title_caption": "TruthfulQA and toxicity evaluations", "what_it_demonstrates": "A sober comparison of behavior metrics with clean percentages.", "source_publication": "GPT‑4 Technical Report", "publication_date": "2023-03-14", "provenance": "Official OpenAI technical report PDF", "exact_or_closest_url": "https://cdn.openai.com/papers/gpt-4.pdf#page=14", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://cdn.openai.com/papers/gpt-4.pdf", "link_precision": "Exact PDF page (14)", "design_tags": "truthfulness; toxicity; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-040", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Table", "visual_type": "Multimodal comparison table", "title_caption": "Audio and speech evaluations", "what_it_demonstrates": "Shows multiple modalities and conditions without icon-heavy decoration.", "source_publication": "GPT‑4o System Card", "publication_date": "2024-08-08", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/gpt-4o-system-card/#model-capabilities", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-4o-system-card/", "link_precision": "Stable system-card section", "design_tags": "audio; multimodal; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-041", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Chart", "visual_type": "Pass-rate comparison chart", "title_caption": "Biological evaluation pass rates", "what_it_demonstrates": "Simple model/baseline comparison with task-family grouping.", "source_publication": "GPT‑4o System Card", "publication_date": "2024-08-08", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/gpt-4o-system-card/#biological-threat-creation", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-4o-system-card/", "link_precision": "Stable system-card section", "design_tags": "bio; pass rate; grouped comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-042", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Table", "visual_type": "Red-team results table", "title_caption": "External red-teaming themes and mitigations", "what_it_demonstrates": "A structured qualitative-quantitative table with grouped risk themes.", "source_publication": "GPT‑4o System Card", "publication_date": "2024-08-08", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/gpt-4o-system-card/#external-red-teaming", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-4o-system-card/", "link_precision": "Stable system-card section", "design_tags": "red teaming; risk themes; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-043", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Chart", "visual_type": "Model-autonomy evaluation chart", "title_caption": "Model autonomy evaluations", "what_it_demonstrates": "Condenses several autonomy tasks into a consistent comparator.", "source_publication": "GPT‑4o System Card", "publication_date": "2024-08-08", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/gpt-4o-system-card/#model-autonomy", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-4o-system-card/", "link_precision": "Stable system-card section", "design_tags": "autonomy; benchmark; model comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-044", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Chart", "visual_type": "Effect-size plot with intervals", "title_caption": "Persuasion effect sizes", "what_it_demonstrates": "A strong forest/dot-plot reference with confidence intervals and a zero line.", "source_publication": "GPT‑4o System Card", "publication_date": "2024-08-08", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/gpt-4o-system-card/#persuasion", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-4o-system-card/", "link_precision": "Stable system-card section", "design_tags": "effect size; confidence intervals; zero line", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-045", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Table", "visual_type": "Preparedness scorecard table", "title_caption": "Preparedness Framework scorecard", "what_it_demonstrates": "A high-stakes categorical matrix with deliberate color restraint.", "source_publication": "GPT‑4o System Card", "publication_date": "2024-08-08", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/gpt-4o-system-card/#preparedness-framework-evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-4o-system-card/", "link_precision": "Stable system-card section", "design_tags": "scorecard; risk; matrix", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-046", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Table", "visual_type": "Safety behavior table", "title_caption": "Refusal and policy-adherence evaluations", "what_it_demonstrates": "Dense safety rates with explicit metric direction and caveats.", "source_publication": "GPT‑4o System Card", "publication_date": "2024-08-08", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/gpt-4o-system-card/#safety-evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-4o-system-card/", "link_precision": "Stable system-card section", "design_tags": "safety table; refusal; policy", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-047", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Chart", "visual_type": "Success-rate chart", "title_caption": "Success rate of GPT‑4o on CTF challenges", "what_it_demonstrates": "Uses graded challenge categories and direct percentages in a sober safety context.", "source_publication": "GPT‑4o System Card", "publication_date": "2024-08-08", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/gpt-4o-system-card/#cybersecurity", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-4o-system-card/", "link_precision": "Stable system-card section", "design_tags": "CTF; success rate; safety", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-048", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark chart", "title_caption": "Autograded Gryphon free response", "what_it_demonstrates": "Simple free-response benchmark bars with direct labels.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/Autograded_Gryphon_Free_Response.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/Autograded_Gryphon_Free_Response.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "free response; benchmark; direct labels", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-049", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark chart", "title_caption": "Biorisk tacit knowledge and troubleshooting", "what_it_demonstrates": "Grouped domain capability comparison with a limited palette.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/Biorisk_Tacit_Knowledge_and_Troubleshooting.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/Biorisk_Tacit_Knowledge_and_Troubleshooting.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "biorisk; tacit knowledge; grouped bars", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-050", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark chart", "title_caption": "GPT‑5 SWE-bench Verified", "what_it_demonstrates": "A clean software benchmark comparison at a fixed sample set.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/GPT-5_SWE-bench_Verified_n%3D477.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/GPT-5_SWE-bench_Verified_n%3D477.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "SWE-bench; coding; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-051", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark chart", "title_caption": "Long-horizon task evaluation", "what_it_demonstrates": "Direct result chart for longer-horizon agentic work.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/horizon2.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/horizon2.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "long horizon; agent; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-052", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Diagram", "visual_type": "Evaluation workflow diagram", "title_caption": "MLE-bench evaluation workflow", "what_it_demonstrates": "Simple technical pipeline with aligned stages and minimal ornament.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/mle.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/mle.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "workflow; benchmark; diagram", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-053", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark chart", "title_caption": "MLE-Bench-30", "what_it_demonstrates": "Compact agent performance chart across ML engineering tasks.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/MLE-Bench-30.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/MLE-Bench-30.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "MLE; agent; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-054", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Example / task figure", "title_caption": "Molecule evaluation example", "what_it_demonstrates": "Restrained presentation of a domain-specific visual task.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/molecule.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/molecule.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "science; example; task visual", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-055", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark chart", "title_caption": "Multimodal troubleshooting and virology", "what_it_demonstrates": "Combines technical troubleshooting and multimodal bio evaluation in a compact card figure.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/Multimodal_Troubleshooting_Virology_Multi-select.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/Multimodal_Troubleshooting_Virology_Multi-select.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "multimodal; bio; troubleshooting", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-056", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark chart", "title_caption": "OpenAI pull requests without browsing", "what_it_demonstrates": "Internal coding comparison with clear conditions.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/OpenAI_PRs_no_browsing.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/OpenAI_PRs_no_browsing.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "coding; internal eval; condition label", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-057", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark chart", "title_caption": "OpenAI-Proof Q&A", "what_it_demonstrates": "Compact advanced-math/science benchmark comparison.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/OpenAI-Proof_Q_A.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/OpenAI-Proof_Q_A.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "proof; Q&A; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-058", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark chart", "title_caption": "PaperBench without browsing", "what_it_demonstrates": "Clear model comparison on research-replication tasks.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/PaperBench_no_browsing.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/PaperBench_no_browsing.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "research replication; benchmark; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-059", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Stacked / categorical bar chart", "title_caption": "Production traffic by deception category", "what_it_demonstrates": "Direct card asset with sparse categories and explicit shares.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/prod_traffic_deception_category.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/prod_traffic_deception_category.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "production traffic; deception; categorical bars", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-060", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark chart", "title_caption": "ProtocolQA open-ended", "what_it_demonstrates": "A clean open-ended scientific protocol comparison.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/ProtocolQA_Open-Ended.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/ProtocolQA_Open-Ended.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "protocol; science; open-ended", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-061", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark chart", "title_caption": "SWE-Lancer independent-contracting software tasks", "what_it_demonstrates": "Direct benchmark comparison suited to coding-agent results.", "source_publication": "GPT‑5 System Card — Deployment Safety Hub", "publication_date": "2025-08-07", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/SWE-Lancer_IC_SWE.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5/assets/SWE-Lancer_IC_SWE.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5", "link_precision": "Direct first-party PNG asset", "design_tags": "coding; benchmark; bar chart", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-062", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Efficiency chart", "title_caption": "CTF performance versus token use", "what_it_demonstrates": "Adds resource consumption to a capability comparison.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image23.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image23.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "CTF; tokens; efficiency", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-063", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark comparison chart", "title_caption": "CTF score comparison", "what_it_demonstrates": "Security benchmark comparison with direct values and a limited palette.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image16.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image16.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "CTF; cyber; benchmark", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-064", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cybersecurity benchmark chart", "title_caption": "CVE exploitation evaluation", "what_it_demonstrates": "A compact domain-specific security result chart.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/cve.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/cve.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "CVE; cybersecurity; benchmark", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-065", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Robustness chart", "title_caption": "CyberGym robustness", "what_it_demonstrates": "Shows robustness under varied conditions using a simple comparator.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image14.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image14.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "CyberGym; robustness; cyber", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-066", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cybersecurity benchmark chart", "title_caption": "ExploitBench", "what_it_demonstrates": "Direct high-resolution comparison on known-vulnerability exploitation.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image6.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image6.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "ExploitBench; cybersecurity; benchmark", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-067", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cybersecurity benchmark chart", "title_caption": "ExploitGym", "what_it_demonstrates": "Matched companion chart for agentic exploitation environments.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image24.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image24.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "ExploitGym; cybersecurity; benchmark", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-068", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost–performance chart", "title_caption": "Internal Research Debugging Eval versus cost", "what_it_demonstrates": "Directly labeled technical frontier with no decorative elements.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image3.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image3.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "debugging; cost-performance; frontier", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-069", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Latency–performance chart", "title_caption": "Internal Research Debugging Eval versus latency", "what_it_demonstrates": "Matched latency companion with shared style and labels.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/debug1.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/debug1.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "debugging; latency-performance; paired plot", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-070", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost–performance chart", "title_caption": "KernelGen 1P versus cost", "what_it_demonstrates": "Technical benchmark frontier suited to systems papers.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image31.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image31.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "kernel generation; cost-performance; frontier", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-071", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Latency–performance chart", "title_caption": "KernelGen 1P versus latency", "what_it_demonstrates": "Matched cost/latency figure pair.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image2.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image2.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "kernel generation; latency-performance; paired plot", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-072", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost–performance chart", "title_caption": "MLE-Bench Revised versus cost", "what_it_demonstrates": "Agentic ML engineering capability against API cost.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image1.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image1.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "MLE-bench; cost-performance; agent", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-073", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Latency–performance chart", "title_caption": "MLE-Bench Revised versus latency", "what_it_demonstrates": "Matched deployment-time comparison for ML engineering agents.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image26.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image26.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "MLE-bench; latency-performance; agent", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-074", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Zoomed cost–performance chart", "title_caption": "NanoGPT cost frontier, zoomed", "what_it_demonstrates": "Shows how to provide a detail view without changing encodings.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image22.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image22.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "zoom; cost-performance; detail view", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-075", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Zoomed latency–performance chart", "title_caption": "NanoGPT latency frontier, zoomed", "what_it_demonstrates": "Matched detail view with the same axes and labels.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image38.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image38.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "zoom; latency-performance; detail view", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-076", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost–performance chart", "title_caption": "NanoGPT versus cost", "what_it_demonstrates": "A clean model/deployment frontier for an end-to-end training task.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image7.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image7.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "NanoGPT; cost-performance; frontier", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-077", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Latency–performance chart", "title_caption": "NanoGPT versus latency", "what_it_demonstrates": "Uses the same grammar to expose response-time trade-offs.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image15.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image15.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "NanoGPT; latency-performance; paired plot", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-078", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Latency–performance chart", "title_caption": "PostTrainBench Lite versus latency", "what_it_demonstrates": "A concise post-training benchmark trade-off.", "source_publication": "GPT‑5.6 preview System Card — Deployment Safety Hub", "publication_date": "2026", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image19.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6-preview/assets/images/image19.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6-preview", "link_precision": "Direct first-party PNG asset", "design_tags": "post-training; latency-performance; benchmark", "notes": "Preview-card asset retained because it exposes the original high-resolution chart directly.", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-079", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Model comparison chart", "title_caption": "Agentic misalignment", "what_it_demonstrates": "A sober categorical safety comparison without alarmist visual treatment.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/Agentic_misalignment.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/Agentic_misalignment.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "misalignment; agentic; safety", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-080", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Small-multiple comparison", "title_caption": "Chain-of-thought controllability by dataset", "what_it_demonstrates": "Shared scales show variation across datasets without legend repetition.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image45.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image45.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "CoT controllability; datasets; small multiples", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-081", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Small-multiple comparison", "title_caption": "Chain-of-thought controllability by instruction", "what_it_demonstrates": "A matched facet design for instruction conditions.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image43.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image43.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "CoT controllability; instructions; facets", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-082", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Small-multiple scatter", "title_caption": "Chain-of-thought monitorability scatterplots", "what_it_demonstrates": "Shared scales and sparse fitted trends across multiple monitoring tasks.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/Scatterplots_main.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/Scatterplots_main.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "monitorability; small multiples; scatter", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-083", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Grouped bar chart", "title_caption": "Chain-of-thought monitorability summary", "what_it_demonstrates": "Compresses several monitorability measures into aligned bars.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/Barplot.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/Barplot.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "monitorability; grouped bars; summary", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-084", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Funnel / scatter chart", "title_caption": "Deployment simulation outcome funnel", "what_it_demonstrates": "Makes filtering and outcome frequency legible without a heavy dashboard.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image48.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image48.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "funnel; deployment; outcomes", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-085", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Scatter plot", "title_caption": "Early versus late metagaming", "what_it_demonstrates": "A direct relationship plot with an interpretable diagonal reference.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image44.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image44.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "metagaming; scatter; diagonal reference", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-086", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Model comparison chart", "title_caption": "GPT‑5.6 versus GPT‑5.5 metagaming", "what_it_demonstrates": "Clean generation-over-generation paired comparison.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image46.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image46.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "metagaming; before-after; paired", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-087", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Model comparison chart", "title_caption": "Hallucination and factuality comparison", "what_it_demonstrates": "Uses limited color and paired conditions for a safety result.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image47.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image47.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "hallucination; factuality; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-088", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Performance–cost scatter", "title_caption": "Hallucination rate versus cost", "what_it_demonstrates": "A compact lower-is-better cost/error frontier.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image34.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image34.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "hallucination; cost; frontier", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-089", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Performance–latency scatter", "title_caption": "Hallucination rate versus latency", "what_it_demonstrates": "Pairs error rate with response time and labels operating points directly.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image37.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image37.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "hallucination; latency; scatter", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-090", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Health benchmark chart", "title_caption": "Health queries requesting patient opinion", "what_it_demonstrates": "A compact, task-specific health safety comparison.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/Health_queries_patient_opinion.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/Health_queries_patient_opinion.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "health; safety; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-091", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Behavior comparison chart", "title_caption": "Impossible tasks", "what_it_demonstrates": "Shows refusal and attempt behavior on impossible tasks with clear categories.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/impossible_tasks.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/impossible_tasks.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "impossible tasks; behavior; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-092", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark figure", "title_caption": "Internal deployment evaluation", "what_it_demonstrates": "Direct internal-evaluation comparison with simple categorical structure.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/internaldep.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/internaldep.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "internal eval; deployment; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-093", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Robustness comparison chart", "title_caption": "Jailbreak robustness", "what_it_demonstrates": "A sparse safety comparison with explicit attack conditions.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image32.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image32.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "jailbreak; robustness; safety", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-094", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Model comparison chart", "title_caption": "Metagaming behavior", "what_it_demonstrates": "A direct comparison of evaluation-aware behavior.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image11.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image11.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "metagaming; behavior; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-095", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Training curve", "title_caption": "Metagaming over training", "what_it_demonstrates": "Uses a simple trajectory to show when behavior emerges.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image8.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image8.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "metagaming; training curve; emergence", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-096", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost–performance chart", "title_caption": "ProtocolQA versus API cost", "what_it_demonstrates": "A clean score/cost frontier for open-ended scientific protocols.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/Protocolqa.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/Protocolqa.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "protocol; cost-performance; frontier", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-097", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Latency–performance chart", "title_caption": "ProtocolQA versus latency", "what_it_demonstrates": "Pairs deployment latency with scientific capability.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image20.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image20.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "protocol; latency-performance; paired plot", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-098", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Behavior comparison chart", "title_caption": "Scruples / moral reasoning", "what_it_demonstrates": "A restrained sensitive-domain comparison with neutral labeling.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/scruples.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/scruples.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "moral reasoning; behavior; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-099", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost–performance chart", "title_caption": "Tacit-knowledge evaluation versus API cost", "what_it_demonstrates": "Sparse score/cost comparison with a focal frontier.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image35.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image35.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "tacit knowledge; cost-performance; frontier", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-100", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Latency–performance chart", "title_caption": "Tacit-knowledge evaluation versus latency", "what_it_demonstrates": "Matched latency companion for an apples-to-apples comparison.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image18.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image18.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "tacit knowledge; latency-performance; paired plot", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-101", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost–performance chart", "title_caption": "Troubleshooting benchmark versus API cost", "what_it_demonstrates": "A deployment-relevant technical benchmark frontier.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image17.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image17.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "troubleshooting; cost-performance; frontier", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-102", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Latency–performance chart", "title_caption": "Troubleshooting benchmark versus latency", "what_it_demonstrates": "Completes a consistent cost/latency pair.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image41.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image41.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "troubleshooting; latency-performance; paired plot", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-103", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost–performance chart", "title_caption": "Virology troubleshooting versus API cost", "what_it_demonstrates": "Directly labeled scientific capability frontier against cost.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image4.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image4.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "bio; cost-performance; frontier", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-104", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Latency–performance chart", "title_caption": "Virology troubleshooting versus latency", "what_it_demonstrates": "Matched companion plot with identical visual grammar.", "source_publication": "GPT‑5.6 System Card — Deployment Safety Hub", "publication_date": "2026-07-09", "provenance": "OpenAI Deployment Safety Hub direct asset", "exact_or_closest_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image5.png", "direct_asset_url": "https://deploymentsafety.openai.com/data/eval-sets/gpt-5-6/assets/images/image5.png", "asset_format": "PNG", "parent_url": "https://deploymentsafety.openai.com/gpt-5-6", "link_precision": "Direct first-party PNG asset", "design_tags": "bio; latency-performance; paired plot", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-105", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Diagram", "visual_type": "Process / efficiency diagram", "title_caption": "Accelerating inference with GPT‑5.6 SOL", "what_it_demonstrates": "Light-theme SVG with exact spacing, sparse annotation, and a clean execution flow.", "source_publication": "GPT‑5.6: Frontier intelligence, more efficiently", "publication_date": "2026-07-09", "provenance": "OpenAI official engineering page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6-frontier-intelligence-efficiency/#accelerating-inference-with-gpt-56-sol", "direct_asset_url": "https://images.ctfassets.net/kftzwdyauwt9/3Hr3nq2SNbbjMbtCGVMbzp/aa96e2f3cdcf9681903d5af69399e240/figure1-light-desktop.svg?w=3840&q=90", "asset_format": "SVG", "parent_url": "https://openai.com/index/gpt-5-6-frontier-intelligence-efficiency/", "link_precision": "Direct SVG asset + stable section anchor", "design_tags": "technical diagram; light theme; SVG; process flow", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-106", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Diagram", "visual_type": "Workflow comparison diagram", "title_caption": "How our agentic harness streamlines repeated work", "what_it_demonstrates": "High-resolution before/after workflow compression and repeated-agent orchestration.", "source_publication": "GPT‑5.6: Frontier intelligence, more efficiently", "publication_date": "2026-07-09", "provenance": "OpenAI official engineering page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6-frontier-intelligence-efficiency/#how-our-agentic-harness-streamlines-repeated-work", "direct_asset_url": "https://images.ctfassets.net/kftzwdyauwt9/4ufC2p4FrIJYDfnieoZ25Y/932a6d7e4931d64792f35f96a3afb1b0/figure3-light-desktop.svg?w=3840&q=90", "asset_format": "SVG", "parent_url": "https://openai.com/index/gpt-5-6-frontier-intelligence-efficiency/", "link_precision": "Direct SVG asset + stable section anchor", "design_tags": "before-after; workflow; SVG; agentic systems", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-107", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Grouped bar chart with error bars", "title_caption": "Figure 10: HealthBench Hard by axis", "what_it_demonstrates": "A matched hard-subset figure with shared visual grammar and clear remaining headroom.", "source_publication": "HealthBench: Evaluating Large Language Models Towards Improved Human Health", "publication_date": "2025-05-13", "provenance": "OpenAI-authored benchmark paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2505.08775v1/figures/hard.png", "direct_asset_url": "https://arxiv.org/html/2505.08775v1/figures/hard.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/healthbench/", "link_precision": "Direct paper image asset", "design_tags": "hard subset; grouped bars; error bars", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-108", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Human–AI comparison chart", "title_caption": "Figure 11: physician-written responses with and without model assistance", "what_it_demonstrates": "Compares expert baselines and augmentation conditions without losing causal grouping.", "source_publication": "HealthBench: Evaluating Large Language Models Towards Improved Human Health", "publication_date": "2025-05-13", "provenance": "OpenAI-authored benchmark paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2505.08775v1#S7", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/healthbench/", "link_precision": "Exact paper section / closest anchor", "design_tags": "human baseline; AI assistance; paired conditions", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-109", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Time-series / milestone chart", "title_caption": "Figure 4: HealthBench performance of OpenAI models over time", "what_it_demonstrates": "Shows model progress as an ordered temporal story rather than a crowded release timeline.", "source_publication": "HealthBench: Evaluating Large Language Models Towards Improved Human Health", "publication_date": "2025-05-13", "provenance": "OpenAI-authored benchmark paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2505.08775v1#S6", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/healthbench/", "link_precision": "Exact paper section / closest anchor", "design_tags": "time series; model progress; milestones", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-110", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Grouped bar chart with error bars", "title_caption": "Figure 5: performance stratified by conversation theme", "what_it_demonstrates": "A dense grouped bar chart that remains interpretable through ordering and dashed overall-score references.", "source_publication": "HealthBench: Evaluating Large Language Models Towards Improved Human Health", "publication_date": "2025-05-13", "provenance": "OpenAI-authored benchmark paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2505.08775v1/figures/theme.png", "direct_asset_url": "https://arxiv.org/html/2505.08775v1/figures/theme.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/healthbench/", "link_precision": "Direct paper image asset", "design_tags": "grouped bars; error bars; themes; overall reference", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-111", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Grouped bar chart with error bars", "title_caption": "Figure 6: performance stratified by axis", "what_it_demonstrates": "Uses the same grammar as the theme chart so readers can compare dimensions without relearning the figure.", "source_publication": "HealthBench: Evaluating Large Language Models Towards Improved Human Health", "publication_date": "2025-05-13", "provenance": "OpenAI-authored benchmark paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2505.08775v1/figures/axis.png", "direct_asset_url": "https://arxiv.org/html/2505.08775v1/figures/axis.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/healthbench/", "link_precision": "Direct paper image asset", "design_tags": "grouped bars; error bars; axes; consistent grammar", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-112", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Score distribution chart", "title_caption": "Figure 8: score distribution over examples by model", "what_it_demonstrates": "Makes benchmark difficulty and saturation visible through distributions rather than averages alone.", "source_publication": "HealthBench: Evaluating Large Language Models Towards Improved Human Health", "publication_date": "2025-05-13", "provenance": "OpenAI-authored benchmark paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2505.08775v1#S6.SS5", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/healthbench/", "link_precision": "Exact paper section / closest anchor", "design_tags": "distribution; benchmark difficulty; saturation", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-113", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Grouped error-rate bar chart", "title_caption": "Figure 9: error rates on consensus criteria by theme", "what_it_demonstrates": "Uses error direction and a dashed overall line to spotlight safety-critical weaknesses.", "source_publication": "HealthBench: Evaluating Large Language Models Towards Improved Human Health", "publication_date": "2025-05-13", "provenance": "OpenAI-authored benchmark paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2505.08775v1/figures/consensus.png", "direct_asset_url": "https://arxiv.org/html/2505.08775v1/figures/consensus.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/healthbench/", "link_precision": "Direct paper image asset", "design_tags": "error rate; grouped bars; safety; error bars", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-114", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Table", "visual_type": "Evaluator reliability table", "title_caption": "Meta-evaluation of model grading", "what_it_demonstrates": "Shows grader agreement and F1 with concise baselines and uncertainty notes.", "source_publication": "HealthBench: Evaluating Large Language Models Towards Improved Human Health", "publication_date": "2025-05-13", "provenance": "OpenAI-authored benchmark paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2505.08775v1#S8", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/healthbench/", "link_precision": "Exact paper section / closest anchor", "design_tags": "grader reliability; meta-evaluation; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-115", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Table", "visual_type": "Pairwise matrix table", "title_caption": "Table 4: pairwise model win rates, including length-controlled comparisons", "what_it_demonstrates": "A clean matrix table with diagonal dashes and consistent percentage precision.", "source_publication": "HealthBench: Evaluating Large Language Models Towards Improved Human Health", "publication_date": "2025-05-13", "provenance": "OpenAI-authored benchmark paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2505.08775v1#S6.SS6", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/healthbench/", "link_precision": "Exact paper section / closest anchor", "design_tags": "matrix table; win rate; length control", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-116", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Table", "visual_type": "Dense benchmark table", "title_caption": "Table 8: 34 HealthBench Consensus criteria across models", "what_it_demonstrates": "Strong reference for a genuinely dense table: grouped criteria, consistent decimals, and explanatory notes.", "source_publication": "HealthBench: Evaluating Large Language Models Towards Improved Human Health", "publication_date": "2025-05-13", "provenance": "OpenAI-authored benchmark paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2505.08775v1#S6.SS7", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/healthbench/", "link_precision": "Exact paper section / closest anchor", "design_tags": "dense table; consensus criteria; grouped rows", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-117", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Safety benchmark table", "title_caption": "Agent safety evaluations", "what_it_demonstrates": "Sober safety table with clear threat classes and mitigation context.", "source_publication": "Introducing ChatGPT agent", "publication_date": "2025-07-17", "provenance": "OpenAI official product/research page", "exact_or_closest_url": "https://openai.com/index/introducing-chatgpt-agent/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-chatgpt-agent/", "link_precision": "Stable evaluations section", "design_tags": "safety; agent; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-118", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Browsing benchmark chart", "title_caption": "BrowseComp", "what_it_demonstrates": "Clean side-by-side comparison of browsing agents.", "source_publication": "Introducing ChatGPT agent", "publication_date": "2025-07-17", "provenance": "OpenAI official product/research page", "exact_or_closest_url": "https://openai.com/index/introducing-chatgpt-agent/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-chatgpt-agent/", "link_precision": "Stable evaluations section", "design_tags": "browsing; agent; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-119", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Benchmark comparison chart", "title_caption": "FrontierMath", "what_it_demonstrates": "Shows difficult-math performance without decorative mathematical motifs.", "source_publication": "Introducing ChatGPT agent", "publication_date": "2025-07-17", "provenance": "OpenAI official product/research page", "exact_or_closest_url": "https://openai.com/index/introducing-chatgpt-agent/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-chatgpt-agent/", "link_precision": "Stable evaluations section", "design_tags": "math; agent; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-120", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Agent benchmark chart", "title_caption": "GAIA", "what_it_demonstrates": "A concise agent benchmark result with model-generation ordering.", "source_publication": "Introducing ChatGPT agent", "publication_date": "2025-07-17", "provenance": "OpenAI official product/research page", "exact_or_closest_url": "https://openai.com/index/introducing-chatgpt-agent/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-chatgpt-agent/", "link_precision": "Stable evaluations section", "design_tags": "agent; benchmark; ordered models", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-121", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Human preference chart", "title_caption": "Human preference on practical tasks", "what_it_demonstrates": "Shows qualitative preference with a clean outcome decomposition.", "source_publication": "Introducing ChatGPT agent", "publication_date": "2025-07-17", "provenance": "OpenAI official product/research page", "exact_or_closest_url": "https://openai.com/index/introducing-chatgpt-agent/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-chatgpt-agent/", "link_precision": "Stable evaluations section", "design_tags": "human preference; stacked bars; practical tasks", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-122", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Benchmark comparison chart", "title_caption": "Humanity’s Last Exam", "what_it_demonstrates": "A compact academic-reasoning comparison with a single focal model.", "source_publication": "Introducing ChatGPT agent", "publication_date": "2025-07-17", "provenance": "OpenAI official product/research page", "exact_or_closest_url": "https://openai.com/index/introducing-chatgpt-agent/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-chatgpt-agent/", "link_precision": "Stable evaluations section", "design_tags": "agent; reasoning; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-123", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Task-family table", "title_caption": "Spreadsheet benchmark results", "what_it_demonstrates": "Grouped quantitative task results with practical labels and clear subtotals.", "source_publication": "Introducing ChatGPT agent", "publication_date": "2025-07-17", "provenance": "OpenAI official product/research page", "exact_or_closest_url": "https://openai.com/index/introducing-chatgpt-agent/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-chatgpt-agent/", "link_precision": "Stable evaluations section", "design_tags": "spreadsheet; task families; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-124", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Computer-use benchmark chart", "title_caption": "WebArena", "what_it_demonstrates": "Uses the same grammar across website-interaction evaluations.", "source_publication": "Introducing ChatGPT agent", "publication_date": "2025-07-17", "provenance": "OpenAI official product/research page", "exact_or_closest_url": "https://openai.com/index/introducing-chatgpt-agent/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-chatgpt-agent/", "link_precision": "Stable evaluations section", "design_tags": "computer use; web; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-125", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Output-length comparison", "title_caption": "Maximum output tokens", "what_it_demonstrates": "A clean before/after product-spec comparison.", "source_publication": "Introducing GPT‑4.1 in the API", "publication_date": "2025-04-14", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/gpt-4-1/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-4-1/", "link_precision": "Stable evaluations or research section", "design_tags": "output length; before-after; specification", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-126", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Instruction-following bar chart", "title_caption": "MultiChallenge", "what_it_demonstrates": "Compact comparison for multi-turn instruction following.", "source_publication": "Introducing GPT‑4.1 in the API", "publication_date": "2025-04-14", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/gpt-4-1/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-4-1/", "link_precision": "Stable evaluations or research section", "design_tags": "instruction following; bar chart; behavior", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-127", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Long-context comparison chart", "title_caption": "OpenAI-MRCR", "what_it_demonstrates": "Compares performance across long-context conditions with a consistent palette.", "source_publication": "Introducing GPT‑4.1 in the API", "publication_date": "2025-04-14", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/gpt-4-1/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-4-1/", "link_precision": "Stable evaluations or research section", "design_tags": "long context; multi-condition; curve", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-128", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Cost / latency table", "title_caption": "Price, latency, and output-token comparison", "what_it_demonstrates": "Puts practical deployment variables next to capability claims with explicit units.", "source_publication": "Introducing GPT‑4.1 in the API", "publication_date": "2025-04-14", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/gpt-4-1/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-4-1/", "link_precision": "Stable evaluations or research section", "design_tags": "pricing; latency; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-129", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Benchmark comparison chart", "title_caption": "SWE-bench Verified", "what_it_demonstrates": "Direct model comparison with an obvious headline result.", "source_publication": "Introducing GPT‑4.1 in the API", "publication_date": "2025-04-14", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/gpt-4-1/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-4-1/", "link_precision": "Stable evaluations or research section", "design_tags": "coding; benchmark; direct values", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-130", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Long-context video chart", "title_caption": "Video long-context understanding", "what_it_demonstrates": "Uses a familiar comparison grammar for a less familiar modality.", "source_publication": "Introducing GPT‑4.1 in the API", "publication_date": "2025-04-14", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/gpt-4-1/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-4-1/", "link_precision": "Stable evaluations or research section", "design_tags": "video; long context; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-131", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Vision benchmark table", "title_caption": "Vision evaluations", "what_it_demonstrates": "Groups several vision tasks without excessive subheaders.", "source_publication": "Introducing GPT‑4.1 in the API", "publication_date": "2025-04-14", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/gpt-4-1/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-4-1/", "link_precision": "Stable evaluations or research section", "design_tags": "vision; benchmark table; grouped rows", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-132", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Human preference comparison", "title_caption": "Comparative evaluations with human testers", "what_it_demonstrates": "Uses paired preference bars and confidence intervals to show qualitative gains.", "source_publication": "Introducing GPT‑4.5", "publication_date": "2025-02-27", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-4-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-4-5/", "link_precision": "Stable evaluations or research section", "design_tags": "human preference; confidence intervals; paired comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-133", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Scaling curve", "title_caption": "Expected loss under unsupervised scaling", "what_it_demonstrates": "A clean line/scatter reference for model scaling without ornamental styling.", "source_publication": "Introducing GPT‑4.5", "publication_date": "2025-02-27", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-4-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-4-5/", "link_precision": "Stable evaluations or research section", "design_tags": "scaling law; curve; log scale", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-134", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Model characteristics table", "title_caption": "Model characteristics", "what_it_demonstrates": "Converts qualitative attributes into a disciplined comparison table.", "source_publication": "Introducing GPT‑4.5", "publication_date": "2025-02-27", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-4-5/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-4-5/", "link_precision": "Stable evaluations or research section", "design_tags": "qualitative table; model traits; compact", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-135", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Benchmark bar chart", "title_caption": "SimpleQA accuracy", "what_it_demonstrates": "Minimal bar comparison with explicit higher-is-better direction.", "source_publication": "Introducing GPT‑4.5", "publication_date": "2025-02-27", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-4-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-4-5/", "link_precision": "Stable evaluations or research section", "design_tags": "factuality; bar chart; simple comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-136", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Error-rate bar chart", "title_caption": "SimpleQA hallucination rate", "what_it_demonstrates": "Pairs the accuracy chart with an inverse metric using matched structure.", "source_publication": "Introducing GPT‑4.5", "publication_date": "2025-02-27", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-4-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-4-5/", "link_precision": "Stable evaluations or research section", "design_tags": "hallucination; paired charts; lower is better", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-137", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Benchmark comparison chart", "title_caption": "Coding", "what_it_demonstrates": "Clean coding comparison with a single accent.", "source_publication": "Introducing GPT‑5", "publication_date": "2025-08-07", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5/", "link_precision": "Stable evaluations or research section", "design_tags": "coding; benchmark; focal model", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-138", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Hallucination / factuality chart", "title_caption": "Factuality and hallucinations", "what_it_demonstrates": "Uses error direction explicitly and avoids conflating accuracy with calibration.", "source_publication": "Introducing GPT‑5", "publication_date": "2025-08-07", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5/", "link_precision": "Stable evaluations or research section", "design_tags": "hallucination; error rate; safety", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-139", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Benchmark comparison chart", "title_caption": "Health", "what_it_demonstrates": "Domain benchmark result with explicit metric and restrained bars.", "source_publication": "Introducing GPT‑5", "publication_date": "2025-08-07", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5/", "link_precision": "Stable evaluations or research section", "design_tags": "health; benchmark; direct values", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-140", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "HealthBench comparison", "title_caption": "HealthBench", "what_it_demonstrates": "A clear single-benchmark model ranking in a high-stakes domain.", "source_publication": "Introducing GPT‑5", "publication_date": "2025-08-07", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5/", "link_precision": "Stable evaluations or research section", "design_tags": "health; benchmark; model ranking", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-141", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Instruction-following chart", "title_caption": "Instruction following", "what_it_demonstrates": "A minimal comparator for behavior and adherence metrics.", "source_publication": "Introducing GPT‑5", "publication_date": "2025-08-07", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5/", "link_precision": "Stable evaluations or research section", "design_tags": "instruction following; behavior; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-142", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Long-context comparison chart", "title_caption": "Long context", "what_it_demonstrates": "Shows performance over context conditions with consistent model ordering.", "source_publication": "Introducing GPT‑5", "publication_date": "2025-08-07", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5/", "link_precision": "Stable evaluations or research section", "design_tags": "long context; small multiples; shared scales", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-143", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Benchmark comparison chart", "title_caption": "Math", "what_it_demonstrates": "Ordered model comparison with a simple higher-is-better reading path.", "source_publication": "Introducing GPT‑5", "publication_date": "2025-08-07", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5/", "link_precision": "Stable evaluations or research section", "design_tags": "math; benchmark; ordered bars", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-144", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Long-context chart", "title_caption": "OpenAI-MRCR", "what_it_demonstrates": "Useful reference for long-context curves and context-length grouping.", "source_publication": "Introducing GPT‑5", "publication_date": "2025-08-07", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5/", "link_precision": "Stable evaluations or research section", "design_tags": "long context; MRCR; curve", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-145", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Safety scorecard table", "title_caption": "Safe-completions and safety evaluations", "what_it_demonstrates": "Pairs capability results with a sober scorecard and compact footnotes.", "source_publication": "Introducing GPT‑5", "publication_date": "2025-08-07", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5/", "link_precision": "Stable evaluations or research section", "design_tags": "safety table; scorecard; footnotes", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-146", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Tool-use comparison chart", "title_caption": "Tool use", "what_it_demonstrates": "Agent/tool performance with concise model labels.", "source_publication": "Introducing GPT‑5", "publication_date": "2025-08-07", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5/", "link_precision": "Stable evaluations or research section", "design_tags": "tool use; agentic; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-147", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Benchmark comparison chart", "title_caption": "Vision", "what_it_demonstrates": "Groups visual reasoning metrics without a rainbow legend.", "source_publication": "Introducing GPT‑5", "publication_date": "2025-08-07", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5/", "link_precision": "Stable evaluations or research section", "design_tags": "vision; benchmark; limited palette", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-148", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Academic benchmark table", "title_caption": "Academic", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.2", "publication_date": "2025-12-11", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-2/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-2/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-149", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Coding benchmark table", "title_caption": "Coding", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.2", "publication_date": "2025-12-11", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-2/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-2/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-150", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Long-context table", "title_caption": "Long context", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.2", "publication_date": "2025-12-11", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-2/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-2/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-151", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost / latency / capability comparison", "title_caption": "Price–performance comparison", "what_it_demonstrates": "Shows a product-level cost/performance claim with few series and direct labels.", "source_publication": "Introducing GPT‑5.2", "publication_date": "2025-12-11", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-2/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5-2/", "link_precision": "Stable evaluations section", "design_tags": "cost-performance; latency; focal model; direct labels", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-152", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Pricing table", "title_caption": "Pricing", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.2", "publication_date": "2025-12-11", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-2/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-2/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-153", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Professional benchmark table", "title_caption": "Professional knowledge work", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.2", "publication_date": "2025-12-11", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-2/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-2/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-154", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Safety benchmark table", "title_caption": "Safety", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.2", "publication_date": "2025-12-11", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-2/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-2/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-155", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Tool-use table", "title_caption": "Tool use", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.2", "publication_date": "2025-12-11", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-2/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-2/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-156", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Vision benchmark table", "title_caption": "Vision", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.2", "publication_date": "2025-12-11", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-2/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-2/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-157", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Abstract-reasoning benchmark table", "title_caption": "Abstract reasoning", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4", "publication_date": "2026-03-05", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-158", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Academic benchmark table", "title_caption": "Academic", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4", "publication_date": "2026-03-05", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-159", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Pricing table", "title_caption": "API pricing", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4", "publication_date": "2026-03-05", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-160", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost / latency / capability comparison", "title_caption": "Capability–latency comparison", "what_it_demonstrates": "Uses a focal model and compact annotation to show an operating-point improvement.", "source_publication": "Introducing GPT‑5.4", "publication_date": "2026-03-05", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5-4/", "link_precision": "Stable evaluations section", "design_tags": "cost-performance; latency; focal model; direct labels", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-161", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Coding benchmark table", "title_caption": "Coding", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4", "publication_date": "2026-03-05", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-162", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Computer-use / vision table", "title_caption": "Computer use and vision", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4", "publication_date": "2026-03-05", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-163", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Long-context benchmark table", "title_caption": "Long context", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4", "publication_date": "2026-03-05", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-164", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "No-reasoning benchmark table", "title_caption": "No-reasoning evaluations", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4", "publication_date": "2026-03-05", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-165", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Professional benchmark table", "title_caption": "Professional", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4", "publication_date": "2026-03-05", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-166", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Compact model-comparison table", "title_caption": "Summary benchmark comparison", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4", "publication_date": "2026-03-05", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-167", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Tool-use benchmark table", "title_caption": "Tool use", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4", "publication_date": "2026-03-05", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-168", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Coding benchmark table", "title_caption": "Coding", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4 mini and nano", "publication_date": "2026-03-17", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-169", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "General-intelligence benchmark table", "title_caption": "Intelligence", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4 mini and nano", "publication_date": "2026-03-17", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-170", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Long-context benchmark table", "title_caption": "Long context", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4 mini and nano", "publication_date": "2026-03-17", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-171", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Multimodal benchmark table", "title_caption": "Multimodal", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4 mini and nano", "publication_date": "2026-03-17", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-172", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Cost / latency table", "title_caption": "Pricing and latency", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4 mini and nano", "publication_date": "2026-03-17", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-173", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost / latency / capability comparison", "title_caption": "Small-model capability / efficiency comparison", "what_it_demonstrates": "A clean template for capability, latency, and price-tier comparisons.", "source_publication": "Introducing GPT‑5.4 mini and nano", "publication_date": "2026-03-17", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", "link_precision": "Stable evaluations section", "design_tags": "cost-performance; latency; focal model; direct labels", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-174", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Compact comparison table", "title_caption": "Summary", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4 mini and nano", "publication_date": "2026-03-17", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-175", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Tool-use benchmark table", "title_caption": "Tool use", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.4 mini and nano", "publication_date": "2026-03-17", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-176", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Academic benchmark table", "title_caption": "Academic", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.5", "publication_date": "2026-04-23", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-5/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-5/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-177", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Coding benchmark table", "title_caption": "Coding", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.5", "publication_date": "2026-04-23", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-5/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-5/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-178", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Computer-use / vision table", "title_caption": "Computer use and vision", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.5", "publication_date": "2026-04-23", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-5/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-5/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-179", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Cybersecurity benchmark table", "title_caption": "Cybersecurity", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.5", "publication_date": "2026-04-23", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-5/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-5/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-180", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Long-context benchmark table", "title_caption": "Long context", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.5", "publication_date": "2026-04-23", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-5/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-5/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-181", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Professional benchmark table", "title_caption": "Professional", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.5", "publication_date": "2026-04-23", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-5/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-5/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-182", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Compact model-comparison table", "title_caption": "Summary benchmark comparison", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.5", "publication_date": "2026-04-23", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-5/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-5/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-183", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Tool-use benchmark table", "title_caption": "Tool use", "what_it_demonstrates": "A disciplined OpenAI launch-table pattern with grouped rows, restrained borders, consistent precision, and sparse emphasis.", "source_publication": "Introducing GPT‑5.5", "publication_date": "2026-04-23", "provenance": "OpenAI official model launch page", "exact_or_closest_url": "https://openai.com/index/introducing-gpt-5-5/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-gpt-5-5/", "link_precision": "Stable evaluations section / nearest table group", "design_tags": "benchmark table; grouped rows; bold best; minimal borders; tabular numerals", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-184", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Academic benchmark table", "title_caption": "Academic", "what_it_demonstrates": "Broad academic results without dashboard styling or excessive decoration.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable evaluations section", "design_tags": "benchmark table; no vertical rules; bold best; grouped rows; compact precision", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-185", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Model-comparison chart", "title_caption": "Artificial Analysis Coding Agent Index", "what_it_demonstrates": "Uses a limited palette and an obvious focal model for a dense multi-model ranking.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#efficient-by-default-maximum-performance-on-demand", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "coding benchmark; multi-model; focal series", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-186", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Automation benchmark chart", "title_caption": "AutomationBench", "what_it_demonstrates": "Compares general-purpose models on realistic workflow automation.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#end-to-end-knowledge-work", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "automation; benchmark; multi-model", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-187", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark chart", "title_caption": "BrowseComp", "what_it_demonstrates": "Orders model generations for immediate reading on a browsing benchmark.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#end-to-end-knowledge-work", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "browsing; benchmark; ordered models", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-188", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Multi-agent comparison", "title_caption": "BrowseComp (Multi-Agent)", "what_it_demonstrates": "Compares single-agent and orchestrated multi-agent performance with one visual grammar.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#end-to-end-knowledge-work", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "multi-agent; before-after; performance scaling", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-189", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cybersecurity benchmark chart", "title_caption": "Capture-the-Flag", "what_it_demonstrates": "Uses graded challenge difficulty and an economical success-rate design.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#pushing-the-frontier-on-cyber-and-science", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "CTF; difficulty; success rate", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-190", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Coding benchmark table", "title_caption": "Coding", "what_it_demonstrates": "Groups related coding evaluations while keeping values easy to scan.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable evaluations section", "design_tags": "benchmark table; no vertical rules; bold best; grouped rows; compact precision", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-191", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Computer-use benchmark table", "title_caption": "Computer use", "what_it_demonstrates": "Wide multi-model comparison that remains readable without vertical rules.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable evaluations section", "design_tags": "benchmark table; no vertical rules; bold best; grouped rows; compact precision", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-192", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Cybersecurity benchmark table", "title_caption": "Cybersecurity", "what_it_demonstrates": "Compact high-stakes capability table with clear row families and sparse emphasis.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable evaluations section", "design_tags": "benchmark table; no vertical rules; bold best; grouped rows; compact precision", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-193", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark comparison chart", "title_caption": "DeepSWE v1.1", "what_it_demonstrates": "Demonstrates concise model ranking with enough context and no dense legend.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#end-to-end-knowledge-work", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "benchmark; ranking; minimal legend", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-194", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cybersecurity benchmark chart", "title_caption": "ExploitBench", "what_it_demonstrates": "Shows security capability with an uncluttered scale and strong baseline contrast.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#pushing-the-frontier-on-cyber-and-science", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "cybersecurity; benchmark; baseline contrast", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-195", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cybersecurity benchmark chart", "title_caption": "ExploitGym", "what_it_demonstrates": "Preserves style across a family of related security evaluations.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#pushing-the-frontier-on-cyber-and-science", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "cybersecurity; consistent series; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-196", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Economic-task benchmark chart", "title_caption": "GDPval-AA v2", "what_it_demonstrates": "Shows expert-grade task performance with restrained categorical comparison.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#end-to-end-knowledge-work", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "economic tasks; expert baseline; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-197", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Science benchmark chart", "title_caption": "GeneBench Pro", "what_it_demonstrates": "A domain-science comparator that avoids decorative biological imagery.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#pushing-the-frontier-on-cyber-and-science", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "science; biology; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-198", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost/performance frontier", "title_caption": "Internal Research Debugging Eval", "what_it_demonstrates": "Shows score against API cost and latency on an internal technical task.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#gpt-5-6-accelerates-openai", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "cost-performance; latency-performance; frontier", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-199", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost/performance frontier", "title_caption": "KernelGen 1P", "what_it_demonstrates": "Technical score/cost comparison with a focused frontier and sparse annotation.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#gpt-5-6-accelerates-openai", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "kernel generation; cost-performance; frontier", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-200", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Science benchmark chart", "title_caption": "LifeSciBench", "what_it_demonstrates": "Scientific benchmark communication with simple ordered categories.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#pushing-the-frontier-on-cyber-and-science", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "life science; benchmark; ordered bars", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-201", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Long-context benchmark table", "title_caption": "Long context", "what_it_demonstrates": "Groups context-length results and uses careful metric labels.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable evaluations section", "design_tags": "benchmark table; no vertical rules; bold best; grouped rows; compact precision", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-202", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Science benchmark chart", "title_caption": "MedChemBench", "what_it_demonstrates": "Keeps a high-density scientific comparison legible in a small footprint.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#pushing-the-frontier-on-cyber-and-science", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "medicinal chemistry; benchmark; compact", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-203", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Multimodal benchmark table", "title_caption": "Multimodal", "what_it_demonstrates": "Mixes image, video, and multimodal metrics in one coherent comparison.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable evaluations section", "design_tags": "benchmark table; no vertical rules; bold best; grouped rows; compact precision", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-204", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cost/performance frontier", "title_caption": "NanoGPT", "what_it_demonstrates": "Reusable score-versus-cost and score-versus-time visual for systems research.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#gpt-5-6-accelerates-openai", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "training systems; cost-performance; latency-performance", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-205", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Computer-use benchmark chart", "title_caption": "OSWorld 2.0", "what_it_demonstrates": "Compact computer-use comparison with consistent naming and direct value labels.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#end-to-end-knowledge-work", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "computer use; benchmark; direct labels", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-206", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Self-improvement benchmark chart", "title_caption": "RSI Index", "what_it_demonstrates": "Frames recursive self-improvement evaluation with a conventional, legible comparator.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#gpt-5-6-accelerates-openai", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "self-improvement; benchmark; model comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-207", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Science and health benchmark table", "title_caption": "Science and health", "what_it_demonstrates": "Handles heterogeneous scientific benchmarks with consistent number formatting.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable evaluations section", "design_tags": "benchmark table; no vertical rules; bold best; grouped rows; compact precision", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-208", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Cybersecurity benchmark chart", "title_caption": "SEC-Bench Pro", "what_it_demonstrates": "Clear model ranking on a difficult, high-stakes domain benchmark.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#pushing-the-frontier-on-cyber-and-science", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "cybersecurity; benchmark; model ranking", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-209", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Multi-agent comparison", "title_caption": "SEC-Bench Pro (Multi-Agent)", "what_it_demonstrates": "Communicates gains from parallel agents without a complicated systems diagram.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#end-to-end-knowledge-work", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "multi-agent; cybersecurity; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-210", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Self-improvement benchmark table", "title_caption": "Self-improvement", "what_it_demonstrates": "Creates hierarchy in a small but high-stakes benchmark family.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable evaluations section", "design_tags": "benchmark table; no vertical rules; bold best; grouped rows; compact precision", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-211", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Benchmark comparison chart", "title_caption": "Terminal-Bench 2.1", "what_it_demonstrates": "A clean categorical agent benchmark comparison with direct values and consistent model order.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#end-to-end-knowledge-work", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "benchmark; coding agents; direct values", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-212", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Multi-agent comparison", "title_caption": "Terminal-Bench 2.1 (Multi-Agent)", "what_it_demonstrates": "Reusable single-versus-multi-agent comparison with aligned scales.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#end-to-end-knowledge-work", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable section anchor", "design_tags": "multi-agent; coding; paired comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-213", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Tool-use benchmark table", "title_caption": "Tool use", "what_it_demonstrates": "Agentic/tool-use metrics with explicit task names and concise values.", "source_publication": "Introducing GPT‑5.6", "publication_date": "2026-07-09", "provenance": "OpenAI official product/research launch page", "exact_or_closest_url": "https://openai.com/index/gpt-5-6/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gpt-5-6/", "link_precision": "Stable evaluations section", "design_tags": "benchmark table; no vertical rules; bold best; grouped rows; compact precision", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-214", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Cost–performance chart", "title_caption": "Cost versus performance", "what_it_demonstrates": "Strong reference for capability plotted against API cost with a visible frontier.", "source_publication": "Introducing OpenAI o3 and o4-mini", "publication_date": "2025-04-16", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-o3-and-o4-mini/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-o3-and-o4-mini/", "link_precision": "Stable evaluations or research section", "design_tags": "cost-performance; frontier; API cost", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-215", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Model comparison table", "title_caption": "Evaluation summary", "what_it_demonstrates": "A compact table for multiple models and reasoning settings.", "source_publication": "Introducing OpenAI o3 and o4-mini", "publication_date": "2025-04-16", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-o3-and-o4-mini/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-o3-and-o4-mini/", "link_precision": "Stable evaluations or research section", "design_tags": "benchmark table; reasoning settings; compact", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-216", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Reasoning-effort comparison", "title_caption": "Low / medium / high reasoning effort", "what_it_demonstrates": "Makes a compute-budget dimension understandable without adding a second legend.", "source_publication": "Introducing OpenAI o3 and o4-mini", "publication_date": "2025-04-16", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-o3-and-o4-mini/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-o3-and-o4-mini/", "link_precision": "Stable evaluations or research section", "design_tags": "reasoning effort; test-time compute; ordered levels", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-217", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Multi-benchmark comparison chart", "title_caption": "Math, coding, and science", "what_it_demonstrates": "Combines several frontier evaluations in a consistent small-multiple grammar.", "source_publication": "Introducing OpenAI o3 and o4-mini", "publication_date": "2025-04-16", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-o3-and-o4-mini/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-o3-and-o4-mini/", "link_precision": "Stable evaluations or research section", "design_tags": "reasoning; multi-benchmark; small multiples", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-218", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Tool-use comparison chart", "title_caption": "Tool use", "what_it_demonstrates": "A sparse chart for multiple tools and agentic tasks.", "source_publication": "Introducing OpenAI o3 and o4-mini", "publication_date": "2025-04-16", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-o3-and-o4-mini/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-o3-and-o4-mini/", "link_precision": "Stable evaluations or research section", "design_tags": "tool use; agentic; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-219", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Visual-reasoning comparison chart", "title_caption": "Visual reasoning", "what_it_demonstrates": "Shows multimodal gains using simple, aligned benchmark panels.", "source_publication": "Introducing OpenAI o3 and o4-mini", "publication_date": "2025-04-16", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/introducing-o3-and-o4-mini/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-o3-and-o4-mini/", "link_precision": "Stable evaluations or research section", "design_tags": "vision; small multiples; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-220", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Accuracy / abstention comparison", "title_caption": "Accuracy and consistency across models", "what_it_demonstrates": "Separates correct, incorrect, and not-attempted outcomes instead of collapsing them.", "source_publication": "Introducing SimpleQA", "publication_date": "2024-10-30", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/introducing-simpleqa/#results", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-simpleqa/", "link_precision": "Stable results/analysis section", "design_tags": "factuality; stacked categories; abstention", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-221", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Consistency comparison", "title_caption": "Accuracy versus consistency", "what_it_demonstrates": "Shows that consistency and correctness are distinct using a compact scatter/comparator.", "source_publication": "Introducing SimpleQA", "publication_date": "2024-10-30", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/introducing-simpleqa/#results", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-simpleqa/", "link_precision": "Stable results/analysis section", "design_tags": "accuracy; consistency; scatter", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-222", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Error analysis figure", "title_caption": "Correct, incorrect, and not attempted", "what_it_demonstrates": "Uses categorical decomposition to make failure modes visible.", "source_publication": "Introducing SimpleQA", "publication_date": "2024-10-30", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/introducing-simpleqa/#analysis", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/introducing-simpleqa/", "link_precision": "Stable results/analysis section", "design_tags": "error decomposition; stacked bars; failure modes", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-223", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Benchmark table", "title_caption": "SimpleQA model results", "what_it_demonstrates": "A small factuality table with clear outcome definitions.", "source_publication": "Introducing SimpleQA", "publication_date": "2024-10-30", "provenance": "OpenAI official benchmark page", "exact_or_closest_url": "https://openai.com/index/introducing-simpleqa/#results", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/introducing-simpleqa/", "link_precision": "Stable results/analysis section", "design_tags": "benchmark table; factuality; compact", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-224", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Efficiency comparison chart", "title_caption": "Jalapeño delivers more throughput per kilowatt", "what_it_demonstrates": "Pairs an engineering efficiency metric with simple hierarchy and strong unit labeling.", "source_publication": "Jalapeño: First results", "publication_date": "2026-08-25", "provenance": "OpenAI official research/engineering page", "exact_or_closest_url": "https://openai.com/index/jalapeno-first-results/#jalapeno-delivers-more-throughput-per-kilowatt", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/jalapeno-first-results/", "link_precision": "Stable section anchor", "design_tags": "energy efficiency; throughput; unit clarity", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-225", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Grouped performance chart", "title_caption": "Jalapeño delivers more tokens per user", "what_it_demonstrates": "Turns infrastructure gains into a user-facing multiplier with minimal decoding effort.", "source_publication": "Jalapeño: First results", "publication_date": "2026-08-25", "provenance": "OpenAI official research/engineering page", "exact_or_closest_url": "https://openai.com/index/jalapeno-first-results/#jalapeno-delivers-more-tokens-per-user", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/jalapeno-first-results/", "link_precision": "Stable section anchor", "design_tags": "big outcome; user-centric metric; direct labels", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-226", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Pareto-frontier chart", "title_caption": "Jalapeño is Pareto frontier at DeepSeek R1 670B", "what_it_demonstrates": "A clean template for communicating system dominance on very large-model inference.", "source_publication": "Jalapeño: First results", "publication_date": "2026-08-25", "provenance": "OpenAI official research/engineering page", "exact_or_closest_url": "https://openai.com/index/jalapeno-first-results/#jalapeno-is-pareto-frontier-at-deepseek-r1-670b", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/jalapeno-first-results/", "link_precision": "Stable section anchor", "design_tags": "pareto frontier; large model; latency", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-227", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Pareto-frontier chart", "title_caption": "Jalapeño is Pareto frontier at GPT-OSS 120B", "what_it_demonstrates": "Highlights a frontier across multiple serving configurations without visual clutter.", "source_publication": "Jalapeño: First results", "publication_date": "2026-08-25", "provenance": "OpenAI official research/engineering page", "exact_or_closest_url": "https://openai.com/index/jalapeno-first-results/#jalapeno-is-pareto-frontier-at-gpt-oss-120b", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/jalapeno-first-results/", "link_precision": "Stable section anchor", "design_tags": "pareto frontier; multi-system; muted competitors", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-228", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Pareto-frontier chart", "title_caption": "Jalapeño is Pareto frontier at Kimi K2.5 1T", "what_it_demonstrates": "Plots extreme-scale systems while keeping the visual calm and legible.", "source_publication": "Jalapeño: First results", "publication_date": "2026-08-25", "provenance": "OpenAI official research/engineering page", "exact_or_closest_url": "https://openai.com/index/jalapeno-first-results/#jalapeno-is-pareto-frontier-at-kimi-k25-1t", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/jalapeno-first-results/", "link_precision": "Stable section anchor", "design_tags": "pareto frontier; extreme scale; calm styling", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-229", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Operating-point comparison", "title_caption": "Jalapeño leads across DeepSeek R1 operating points", "what_it_demonstrates": "Repeats the same visual grammar so cross-panel comparison remains effortless.", "source_publication": "Jalapeño: First results", "publication_date": "2026-08-25", "provenance": "OpenAI official research/engineering page", "exact_or_closest_url": "https://openai.com/index/jalapeno-first-results/#jalapeno-leads-across-deepseek-r1-operating-points", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/jalapeno-first-results/", "link_precision": "Stable section anchor", "design_tags": "repeatable grammar; small multiples; operating points", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-230", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Operating-point comparison", "title_caption": "Jalapeño leads across GPT-OSS operating points", "what_it_demonstrates": "Compares a system across several practically meaningful operating points with consistent scales.", "source_publication": "Jalapeño: First results", "publication_date": "2026-08-25", "provenance": "OpenAI official research/engineering page", "exact_or_closest_url": "https://openai.com/index/jalapeno-first-results/#jalapeno-leads-across-gpt-oss-operating-points", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/jalapeno-first-results/", "link_precision": "Stable section anchor", "design_tags": "operating points; small multiples; consistent scale", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-231", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Operating-point comparison", "title_caption": "Jalapeño leads across Kimi K2.5 operating points", "what_it_demonstrates": "Completes a coherent visual series in which only data and labels change, not the grammar.", "source_publication": "Jalapeño: First results", "publication_date": "2026-08-25", "provenance": "OpenAI official research/engineering page", "exact_or_closest_url": "https://openai.com/index/jalapeno-first-results/#jalapeno-leads-across-kimi-k25-operating-points", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/jalapeno-first-results/", "link_precision": "Stable section anchor", "design_tags": "series consistency; operating points; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-232", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Benchmark comparison chart", "title_caption": "AIME, Codeforces, and GPQA", "what_it_demonstrates": "Presents three very different benchmarks with a shared visual grammar.", "source_publication": "Learning to reason with LLMs", "publication_date": "2024-09-12", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/learning-to-reason-with-llms/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/learning-to-reason-with-llms/", "link_precision": "Stable evaluations or research section", "design_tags": "reasoning; multi-benchmark; small multiples", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-233", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Agent success curve", "title_caption": "CTF agent performance", "what_it_demonstrates": "A restrained success-rate curve over challenge settings.", "source_publication": "Learning to reason with LLMs", "publication_date": "2024-09-12", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/learning-to-reason-with-llms/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/learning-to-reason-with-llms/", "link_precision": "Stable evaluations or research section", "design_tags": "agent; CTF; success rate", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-234", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Pass@k / consensus chart", "title_caption": "Performance with consensus among multiple samples", "what_it_demonstrates": "Shows inference-time aggregation without obscuring the single-sample baseline.", "source_publication": "Learning to reason with LLMs", "publication_date": "2024-09-12", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/learning-to-reason-with-llms/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/learning-to-reason-with-llms/", "link_precision": "Stable evaluations or research section", "design_tags": "consensus; pass@k; test-time compute", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-235", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Human preference chart", "title_caption": "Preference on difficult prompts", "what_it_demonstrates": "Pairs a simple preference statistic with careful scope notes.", "source_publication": "Learning to reason with LLMs", "publication_date": "2024-09-12", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/learning-to-reason-with-llms/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/learning-to-reason-with-llms/", "link_precision": "Stable evaluations or research section", "design_tags": "human preference; paired comparison; confidence", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-236", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Percentile / distribution chart", "title_caption": "Programming competition performance", "what_it_demonstrates": "Converts raw score into an intuitive competitive percentile.", "source_publication": "Learning to reason with LLMs", "publication_date": "2024-09-12", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/learning-to-reason-with-llms/#evaluations", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/learning-to-reason-with-llms/", "link_precision": "Stable evaluations or research section", "design_tags": "percentile; coding; distribution", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-237", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart / Table", "visual_type": "Safety / preparedness table", "title_caption": "Safety evaluations", "what_it_demonstrates": "A sober model-card table with explicit categories and caveats.", "source_publication": "Learning to reason with LLMs", "publication_date": "2024-09-12", "provenance": "OpenAI official model/research page", "exact_or_closest_url": "https://openai.com/index/learning-to-reason-with-llms/#evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/learning-to-reason-with-llms/", "link_precision": "Stable evaluations or research section", "design_tags": "safety; scorecard; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-238", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Time / cost comparison", "title_caption": "Expert time and model completion time", "what_it_demonstrates": "Pairs quality with practical completion-time or cost context.", "source_publication": "Measuring the performance of our models on real-world tasks (GDPval)", "publication_date": "2025-09-25", "provenance": "OpenAI official economic research page", "exact_or_closest_url": "https://openai.com/index/gdpval/#early-results", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gdpval/", "link_precision": "Stable section anchor", "design_tags": "time; cost; productivity", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-239", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Diagram", "visual_type": "Benchmark construction diagram", "title_caption": "GDPval task and deliverable workflow", "what_it_demonstrates": "Explains prompt, reference files, model deliverable, and expert grading in one flow.", "source_publication": "Measuring the performance of our models on real-world tasks (GDPval)", "publication_date": "2025-09-25", "provenance": "OpenAI official economic research page", "exact_or_closest_url": "https://openai.com/index/gdpval/#how-we-grade-model-performance", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gdpval/", "link_precision": "Stable section anchor", "design_tags": "workflow; benchmark; grading", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-240", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Table", "visual_type": "Expert-preference results table", "title_caption": "Model versus expert win-rate table", "what_it_demonstrates": "Compact percentages with clear outcome definitions and denominator notes.", "source_publication": "Measuring the performance of our models on real-world tasks (GDPval)", "publication_date": "2025-09-25", "provenance": "OpenAI official economic research page", "exact_or_closest_url": "https://openai.com/index/gdpval/#early-results", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/gdpval/", "link_precision": "Stable section anchor", "design_tags": "table; expert preference; percentages", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-241", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Grouped comparison chart", "title_caption": "Performance by industry", "what_it_demonstrates": "A clean hierarchical roll-up from detailed occupations to broader sectors.", "source_publication": "Measuring the performance of our models on real-world tasks (GDPval)", "publication_date": "2025-09-25", "provenance": "OpenAI official economic research page", "exact_or_closest_url": "https://openai.com/index/gdpval/#early-results", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gdpval/", "link_precision": "Stable section anchor", "design_tags": "industry; grouped bars; hierarchy", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-242", "recommended_rank": "", "quality_grade": "A — strong", "reference_tier": "1 — Current OpenAI web / Safety Hub", "category": "Chart", "visual_type": "Small-multiple comparison", "title_caption": "Performance by occupation", "what_it_demonstrates": "Shows heterogeneous performance across occupations while preserving a shared reading order.", "source_publication": "Measuring the performance of our models on real-world tasks (GDPval)", "publication_date": "2025-09-25", "provenance": "OpenAI official economic research page", "exact_or_closest_url": "https://openai.com/index/gdpval/#early-results", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/gdpval/", "link_precision": "Stable section anchor", "design_tags": "small multiples; occupations; shared scale", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-243", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Table", "visual_type": "Ablation table", "title_caption": "Agent scaffolding and retry ablations", "what_it_demonstrates": "A clean intervention table with one factor per row and aggregate outcomes.", "source_publication": "MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering", "publication_date": "2024-10-09", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2410.07095#page=12", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/mle-bench/", "link_precision": "Exact PDF page", "design_tags": "ablation; scaffolding; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-244", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Diagram", "visual_type": "Benchmark pipeline diagram", "title_caption": "Figure 1: MLE-bench overview", "what_it_demonstrates": "Shows task, agent loop, submission, and Kaggle-style evaluation in one compact schematic.", "source_publication": "MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering", "publication_date": "2024-10-09", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2410.07095v6/x1.png", "direct_asset_url": "https://arxiv.org/html/2410.07095v6/x1.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/mle-bench/", "link_precision": "Direct paper image asset", "design_tags": "pipeline; MLE; diagram", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-245", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Rank / medal chart", "title_caption": "Figure 2: medals by competition percentile", "what_it_demonstrates": "Maps agent score into an intuitive competition percentile and medal tier.", "source_publication": "MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering", "publication_date": "2024-10-09", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2410.07095v6/x2.png", "direct_asset_url": "https://arxiv.org/html/2410.07095v6/x2.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/mle-bench/", "link_precision": "Direct paper image asset", "design_tags": "percentile; medals; competition", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-246", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Scaling curve", "title_caption": "Figure 3: percentage of competitions with medals versus attempts", "what_it_demonstrates": "Shows gains from repeated attempts with a simple monotonic curve.", "source_publication": "MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering", "publication_date": "2024-10-09", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2410.07095#page=7", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/mle-bench/", "link_precision": "Exact PDF page", "design_tags": "attempts; test-time compute; scaling", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-247", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Pass@k curve", "title_caption": "Figure 4: pass@k / medal acquisition", "what_it_demonstrates": "A direct reference for agent performance as sample count increases.", "source_publication": "MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering", "publication_date": "2024-10-09", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2410.07095#page=8", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/mle-bench/", "link_precision": "Exact PDF page", "design_tags": "pass@k; agent; curve", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-248", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Scatter / grouped chart", "title_caption": "Figure 5: performance versus competition difficulty", "what_it_demonstrates": "Makes heterogeneity across easy and hard competitions visible.", "source_publication": "MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering", "publication_date": "2024-10-09", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2410.07095#page=8", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/mle-bench/", "link_precision": "Exact PDF page", "design_tags": "difficulty; performance; scatter", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-249", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Contamination analysis chart", "title_caption": "Figure 6: performance and possible contamination", "what_it_demonstrates": "A compact diagnostic plot separating benchmark validity from raw score.", "source_publication": "MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering", "publication_date": "2024-10-09", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2410.07095#page=10", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/mle-bench/", "link_precision": "Exact PDF page", "design_tags": "contamination; diagnostic; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-250", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Table", "visual_type": "Benchmark table", "title_caption": "Main MLE-bench medal results", "what_it_demonstrates": "Rows agents and conditions against clear aggregate medal outcomes.", "source_publication": "MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering", "publication_date": "2024-10-09", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2410.07095#page=6", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/mle-bench/", "link_precision": "Exact PDF page", "design_tags": "benchmark table; medals; agents", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-251", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Chart", "visual_type": "Agent benchmark chart", "title_caption": "Agent performance on test suite", "what_it_demonstrates": "A clean success-rate comparison across agent tasks.", "source_publication": "OpenAI o1 System Card", "publication_date": "2024-12-05", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/openai-o1-system-card/#model-autonomy", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/openai-o1-system-card/", "link_precision": "Stable system-card section", "design_tags": "agent; success rate; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-252", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Chart", "visual_type": "Biological-risk comparison", "title_caption": "Biological threat evaluations", "what_it_demonstrates": "Grouped pass rates with explicit baselines.", "source_publication": "OpenAI o1 System Card", "publication_date": "2024-12-05", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/openai-o1-system-card/#biological-threat-creation", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/openai-o1-system-card/", "link_precision": "Stable system-card section", "design_tags": "bio; risk; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-253", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Table", "visual_type": "Safety behavior table", "title_caption": "Disallowed-content and over-refusal metrics", "what_it_demonstrates": "Pairs opposing safety metrics to prevent one-sided interpretation.", "source_publication": "OpenAI o1 System Card", "publication_date": "2024-12-05", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/openai-o1-system-card/#safety-evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/openai-o1-system-card/", "link_precision": "Stable system-card section", "design_tags": "safety; trade-off; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-254", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Table", "visual_type": "Jailbreak robustness table", "title_caption": "Jailbreak and refusal evaluations", "what_it_demonstrates": "Consistent rates and conditions in a wide safety table.", "source_publication": "OpenAI o1 System Card", "publication_date": "2024-12-05", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/openai-o1-system-card/#safety-evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/openai-o1-system-card/", "link_precision": "Stable system-card section", "design_tags": "jailbreak; refusal; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-255", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Table", "visual_type": "Autonomy benchmark table", "title_caption": "Model autonomy evaluations", "what_it_demonstrates": "A compact multi-task autonomy result table.", "source_publication": "OpenAI o1 System Card", "publication_date": "2024-12-05", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/openai-o1-system-card/#model-autonomy", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/openai-o1-system-card/", "link_precision": "Stable system-card section", "design_tags": "autonomy; benchmark table; tasks", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-256", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Table", "visual_type": "Effect-size / preference table", "title_caption": "Persuasion evaluations", "what_it_demonstrates": "Combines effect magnitude and uncertainty in a disciplined report table.", "source_publication": "OpenAI o1 System Card", "publication_date": "2024-12-05", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/openai-o1-system-card/#persuasion", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/openai-o1-system-card/", "link_precision": "Stable system-card section", "design_tags": "persuasion; effect size; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-257", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Table", "visual_type": "Preparedness scorecard table", "title_caption": "Preparedness Framework scorecard", "what_it_demonstrates": "A compact risk matrix that makes threshold status immediately visible.", "source_publication": "OpenAI o1 System Card", "publication_date": "2024-12-05", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/openai-o1-system-card/#preparedness-framework-evaluations", "direct_asset_url": "", "asset_format": "HTML / PDF table", "parent_url": "https://openai.com/index/openai-o1-system-card/", "link_precision": "Stable system-card section", "design_tags": "scorecard; risk; matrix", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-258", "recommended_rank": "", "quality_grade": "A− — report-grade", "reference_tier": "2 — OpenAI technical/system report", "category": "Chart", "visual_type": "Success-rate chart", "title_caption": "Success rate of o1 on CTF challenges", "what_it_demonstrates": "Uses task difficulty and model variants without an overloaded legend.", "source_publication": "OpenAI o1 System Card", "publication_date": "2024-12-05", "provenance": "OpenAI official system card page", "exact_or_closest_url": "https://openai.com/index/openai-o1-system-card/#cybersecurity", "direct_asset_url": "", "asset_format": "Interactive / HTML", "parent_url": "https://openai.com/index/openai-o1-system-card/", "link_precision": "Stable system-card section", "design_tags": "CTF; success rate; model variants", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-259", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Diagram", "visual_type": "Benchmark pipeline diagram", "title_caption": "Figure 1: PaperBench overview", "what_it_demonstrates": "A clean end-to-end benchmark diagram covering paper, agent, reproduction, and judge.", "source_publication": "PaperBench: Evaluating AI’s Ability to Replicate AI Research", "publication_date": "2025-04-02", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2504.01848v3/x1.png", "direct_asset_url": "https://arxiv.org/html/2504.01848v3/x1.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/paperbench/", "link_precision": "Direct paper image asset", "design_tags": "pipeline; research replication; diagram", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-260", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Human–agent comparison chart", "title_caption": "Figure 3: human versus agent performance", "what_it_demonstrates": "A direct comparison against expert human baselines with aligned task groups.", "source_publication": "PaperBench: Evaluating AI’s Ability to Replicate AI Research", "publication_date": "2025-04-02", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2504.01848#page=7", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/paperbench/", "link_precision": "Exact PDF page", "design_tags": "human baseline; agent; comparison", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-261", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Progress curve", "title_caption": "Figure 4a: incremental progress on a lower-cost paper", "what_it_demonstrates": "Shows work accumulation over time with one simple trajectory per system.", "source_publication": "PaperBench: Evaluating AI’s Ability to Replicate AI Research", "publication_date": "2025-04-02", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2504.01848v3/x8.png", "direct_asset_url": "https://arxiv.org/html/2504.01848v3/x8.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/paperbench/", "link_precision": "Direct paper image asset", "design_tags": "progress curve; time; agent", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-262", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Progress curve", "title_caption": "Figure 4b: incremental progress on a higher-cost paper", "what_it_demonstrates": "Matched companion plot that enables apples-to-apples comparison.", "source_publication": "PaperBench: Evaluating AI’s Ability to Replicate AI Research", "publication_date": "2025-04-02", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2504.01848v3/x9.png", "direct_asset_url": "https://arxiv.org/html/2504.01848v3/x9.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/paperbench/", "link_precision": "Direct paper image asset", "design_tags": "progress curve; time; paired figure", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-263", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Distribution chart", "title_caption": "Figure 5: subset test-score distribution", "what_it_demonstrates": "Shows evaluator discrimination and benchmark spread beyond a single mean.", "source_publication": "PaperBench: Evaluating AI’s Ability to Replicate AI Research", "publication_date": "2025-04-02", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2504.01848#page=16", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/paperbench/", "link_precision": "Exact PDF page", "design_tags": "distribution; judge; benchmark", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-264", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Correlation scatter", "title_caption": "Figure 7: JudgeEval versus JudgeEvalG F1", "what_it_demonstrates": "A simple agreement/correlation figure with a clear diagonal reading.", "source_publication": "PaperBench: Evaluating AI’s Ability to Replicate AI Research", "publication_date": "2025-04-02", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2504.01848#page=17", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/paperbench/", "link_precision": "Exact PDF page", "design_tags": "correlation; judge; scatter", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-265", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Table", "visual_type": "Evaluator comparison table", "title_caption": "JudgeEval results table", "what_it_demonstrates": "Side-by-side evaluator quality and cost metrics with consistent precision.", "source_publication": "PaperBench: Evaluating AI’s Ability to Replicate AI Research", "publication_date": "2025-04-02", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2504.01848#page=16", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/paperbench/", "link_precision": "Exact PDF page", "design_tags": "evaluator; cost; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-266", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Table", "visual_type": "Benchmark table", "title_caption": "Main agent performance table", "what_it_demonstrates": "Dense agent results with task subsets, conditions, and clear aggregate columns.", "source_publication": "PaperBench: Evaluating AI’s Ability to Replicate AI Research", "publication_date": "2025-04-02", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2504.01848#page=8", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/paperbench/", "link_precision": "Exact PDF page", "design_tags": "benchmark table; agents; aggregates", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-267", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Table", "visual_type": "Ablation table", "title_caption": "PaperBench agent ablations", "what_it_demonstrates": "A clean ablation layout that isolates one intervention per row.", "source_publication": "PaperBench: Evaluating AI’s Ability to Replicate AI Research", "publication_date": "2025-04-02", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2504.01848#page=19", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/paperbench/", "link_precision": "Exact PDF page", "design_tags": "ablation; agent; table", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-268", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Diagram", "visual_type": "Benchmark overview diagram", "title_caption": "Figure 1: SWE-Lancer task and evaluation overview", "what_it_demonstrates": "Explains issue tasks, bounty value, model attempt, and judging in one flow.", "source_publication": "SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?", "publication_date": "2025-02-17", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2502.12115v3/x1.png", "direct_asset_url": "https://arxiv.org/html/2502.12115v3/x1.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/swe-lancer/", "link_precision": "Direct paper image asset", "design_tags": "workflow; freelance tasks; diagram", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-269", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Task distribution chart", "title_caption": "Figure 2: task types and values", "what_it_demonstrates": "Shows dataset composition using a compact distribution and monetary context.", "source_publication": "SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?", "publication_date": "2025-02-17", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2502.12115v3/x2.png", "direct_asset_url": "https://arxiv.org/html/2502.12115v3/x2.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/swe-lancer/", "link_precision": "Direct paper image asset", "design_tags": "dataset; task types; value", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-270", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Grouped bar chart", "title_caption": "Figure 3: resolved tasks by price bucket", "what_it_demonstrates": "Combines task difficulty and value buckets with model success rates.", "source_publication": "SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?", "publication_date": "2025-02-17", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2502.12115v3/x3.png", "direct_asset_url": "https://arxiv.org/html/2502.12115v3/x3.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/swe-lancer/", "link_precision": "Direct paper image asset", "design_tags": "price buckets; success rate; grouped bars", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-271", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Value comparison chart", "title_caption": "Figure 5: realized value by model", "what_it_demonstrates": "Converts benchmark success into an intuitive dollar-value outcome.", "source_publication": "SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?", "publication_date": "2025-02-17", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2502.12115v3/x5.png", "direct_asset_url": "https://arxiv.org/html/2502.12115v3/x5.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/swe-lancer/", "link_precision": "Direct paper image asset", "design_tags": "economic value; model comparison; bars", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-272", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Cost–value chart", "title_caption": "Figure 6: model cost versus task value", "what_it_demonstrates": "A deployment-oriented cost/value comparison with clear units.", "source_publication": "SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?", "publication_date": "2025-02-17", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2502.12115v3/x6.png", "direct_asset_url": "https://arxiv.org/html/2502.12115v3/x6.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/swe-lancer/", "link_precision": "Direct paper image asset", "design_tags": "cost-value; scatter; economic tasks", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-273", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Outcome decomposition chart", "title_caption": "Figure 7: accepted, rejected, and unresolved outcomes", "what_it_demonstrates": "Separates outcomes instead of reporting only aggregate success.", "source_publication": "SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?", "publication_date": "2025-02-17", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2502.12115v3/x7.png", "direct_asset_url": "https://arxiv.org/html/2502.12115v3/x7.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/swe-lancer/", "link_precision": "Direct paper image asset", "design_tags": "outcomes; stacked bars; failure modes", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-274", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Chart", "visual_type": "Before/after value chart", "title_caption": "Figure 8: expected realized value and cost savings", "what_it_demonstrates": "Makes the economic implication of performance visible with a simple comparison.", "source_publication": "SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?", "publication_date": "2025-02-17", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2502.12115v3/x8.png", "direct_asset_url": "https://arxiv.org/html/2502.12115v3/x8.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/swe-lancer/", "link_precision": "Direct paper image asset", "design_tags": "cost savings; before-after; economic value", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-275", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Table", "visual_type": "Benchmark table", "title_caption": "Main SWE-Lancer results", "what_it_demonstrates": "Combines resolved tasks, bounty value, and cost in a compact results table.", "source_publication": "SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?", "publication_date": "2025-02-17", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/pdf/2502.12115#page=5", "direct_asset_url": "", "asset_format": "PDF page", "parent_url": "https://openai.com/index/swe-lancer/", "link_precision": "Exact PDF page", "design_tags": "benchmark table; value; cost", "notes": "", "accessed_date": "2026-09-02"}
{"id": "OAI-VIS-276", "recommended_rank": "", "quality_grade": "B+ — research-grade", "reference_tier": "3 — OpenAI-authored research paper", "category": "Table", "visual_type": "Task-category table", "title_caption": "Results by task category and price bucket", "what_it_demonstrates": "A grouped breakdown table with natural row hierarchy.", "source_publication": "SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?", "publication_date": "2025-02-17", "provenance": "OpenAI-authored research paper + official OpenAI landing page", "exact_or_closest_url": "https://arxiv.org/html/2502.12115v3/x9.png", "direct_asset_url": "https://arxiv.org/html/2502.12115v3/x9.png", "asset_format": "PNG", "parent_url": "https://openai.com/index/swe-lancer/", "link_precision": "Direct paper image asset", "design_tags": "task categories; price buckets; table", "notes": "", "accessed_date": "2026-09-02"}

SHA-256: fb04dc85ec9227dc6fa6780e368a03b29c38a7c5db9a6792acadbc4705403989