{"source":"BenchGecko","url":"https://benchgecko.ai/model/o3","as_of":"2026-10-05","license":"BenchGecko collected data (prices, provider offers, Gecko Tests) is CC BY 4.0 · https://creativecommons.org/licenses/by/4.0/. Benchmark scores belong to their original publishers (see sources) and are aggregated with attribution. Attribution required: \"Source: BenchGecko\" with a link.","attribution":"Source: BenchGecko · https://benchgecko.ai/model/o3","cite":"o3 · benchmarks, pricing and providers. BenchGecko, data as of 2026-10-05. https://benchgecko.ai/model/o3","slug":"o3","name":"o3","provider":{"name":"OpenAI","slug":"openai"},"model_type":"multimodal","release_date":"2025-04-16","is_open_source":false,"status":"active","context_window":200000,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","benchgecko_score":{"value":55.1,"rank":118,"of":312,"method":"Normalized average of public benchmark scores","method_url":"https://benchgecko.ai/methodology"},"avg_score":55.1,"list_price":{"input_usd_per_m":2,"output_usd_per_m":8,"source":"OpenRouter models API (checked daily)","as_of":"2026-10-05"},"pricing":{"input":2,"output":8},"providers":[{"provider":"OpenAI","endpoint":"openai","quantization":null,"context_tokens":200000,"input_usd_per_m":2,"output_usd_per_m":8,"cache_read_usd_per_m":0.5,"uptime_1d_pct":100,"as_of":"2026-10-05"}],"cheapest_provider":{"provider":"OpenAI","endpoint":"openai","quantization":null,"context_tokens":200000,"input_usd_per_m":2,"output_usd_per_m":8,"cache_read_usd_per_m":0.5,"uptime_1d_pct":100,"as_of":"2026-10-05"},"providers_source":"OpenRouter endpoints API, one row per provider, refreshed daily","scores":[{"benchmark":"MATH level 5","benchmark_slug":"math-level-5","score":97.77,"unit":"%","category":"math","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/math-level-5"},{"benchmark":"Fiction.LiveBench","benchmark_slug":"fiction-livebench","score":88.9,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/fiction-livebench"},{"benchmark":"HELM — IFEval","benchmark_slug":"helm-ifeval","score":86.9,"unit":"%","category":"language","source":"Stanford HELM","source_url":"https://crfm.stanford.edu/helm/","benchmark_url":"https://benchgecko.ai/benchmark/helm-ifeval"},{"benchmark":"HELM — WildBench","benchmark_slug":"helm-wildbench","score":86.1,"unit":"%","category":"reasoning","source":"Stanford HELM","source_url":"https://crfm.stanford.edu/helm/","benchmark_url":"https://benchgecko.ai/benchmark/helm-wildbench"},{"benchmark":"HELM — MMLU-Pro","benchmark_slug":"helm-mmlu-pro","score":85.9,"unit":"%","category":"knowledge","source":"Stanford HELM","source_url":"https://crfm.stanford.edu/helm/","benchmark_url":"https://benchgecko.ai/benchmark/helm-mmlu-pro"},{"benchmark":"OTIS Mock AIME 2024-2025","benchmark_slug":"otis-mock-aime-2024-2025","score":84.43,"unit":"%","category":"math","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/otis-mock-aime-2024-2025"},{"benchmark":"Lech Mazur Writing","benchmark_slug":"lech-mazur-writing","score":83.9,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/lech-mazur-writing"},{"benchmark":"Artificial Analysis · GPQA Diamond","benchmark_slug":"aa-gpqa-diamond","score":82.7,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-gpqa-diamond"},{"benchmark":"Aider polyglot","benchmark_slug":"aider-polyglot","score":81.3,"unit":"%","category":"coding","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/aider-polyglot"},{"benchmark":"Artificial Analysis · tau2-Bench Telecom","benchmark_slug":"aa-tau2-bench","score":80.7,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-tau2-bench"},{"benchmark":"GPQA diamond","benchmark_slug":"gpqa-diamond","score":75.76,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/gpqa-diamond"},{"benchmark":"HELM — GPQA","benchmark_slug":"helm-gpqa","score":75.3,"unit":"%","category":"knowledge","source":"Stanford HELM","source_url":"https://crfm.stanford.edu/helm/","benchmark_url":"https://benchgecko.ai/benchmark/helm-gpqa"},{"benchmark":"Artificial Analysis · Long Context Reasoning","benchmark_slug":"aa-long-context-reasoning","score":74.7,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-long-context-reasoning"},{"benchmark":"Dtbench","benchmark_slug":"dtbench","score":74.67,"unit":"%","category":"general","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/dtbench"},{"benchmark":"GeoBench","benchmark_slug":"geobench","score":74,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/geobench"},{"benchmark":"CadEval","benchmark_slug":"cadeval","score":74,"unit":"%","category":"coding","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/cadeval"},{"benchmark":"HELM — Omni-MATH","benchmark_slug":"helm-omni-math","score":71.4,"unit":"%","category":"math","source":"Stanford HELM","source_url":"https://crfm.stanford.edu/helm/","benchmark_url":"https://benchgecko.ai/benchmark/helm-omni-math"},{"benchmark":"Artificial Analysis · IFBench","benchmark_slug":"aa-ifbench","score":71.4,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-ifbench"},{"benchmark":"Artificial Analysis · MMMU Pro","benchmark_slug":"aa-mmmu-pro","score":70.1,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-mmmu-pro"},{"benchmark":"Metr Time Horizons","benchmark_slug":"metr-time-horizons","score":65.44,"unit":"%","category":"general","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/metr-time-horizons"},{"benchmark":"SWE-Bench verified","benchmark_slug":"swe-bench-verified","score":62.32,"unit":"%","category":"coding","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/swe-bench-verified"},{"benchmark":"ARC-AGI","benchmark_slug":"arc-agi","score":60.8,"unit":"%","category":"reasoning","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/arc-agi"},{"benchmark":"SWE-Bench Verified (Bash Only)","benchmark_slug":"swe-bench-verified-bash-only","score":58.4,"unit":"%","category":"coding","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/swe-bench-verified-bash-only"},{"benchmark":"SimpleQA Verified","benchmark_slug":"simpleqa-verified","score":53,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/simpleqa-verified"},{"benchmark":"WeirdML","benchmark_slug":"weirdml","score":52.42,"unit":"%","category":"coding","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/weirdml"},{"benchmark":"Professional Reasoning — Legal","benchmark_slug":"seal-pro-reasoning-legal","score":48.57,"unit":"%","category":"knowledge","source":"Scale SEAL leaderboards","source_url":"https://scale.com/leaderboard","benchmark_url":"https://benchgecko.ai/benchmark/seal-pro-reasoning-legal"},{"benchmark":"Professional Reasoning — Finance","benchmark_slug":"seal-pro-reasoning-finance","score":47.69,"unit":"%","category":"knowledge","source":"Scale SEAL leaderboards","source_url":"https://scale.com/leaderboard","benchmark_url":"https://benchgecko.ai/benchmark/seal-pro-reasoning-finance"},{"benchmark":"Lmca","benchmark_slug":"lmca","score":46.71,"unit":"%","category":"general","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/lmca"},{"benchmark":"DeepResearch Bench","benchmark_slug":"deepresearch-bench","score":45.2,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/deepresearch-bench"},{"benchmark":"SimpleBench","benchmark_slug":"simplebench","score":43.72,"unit":"%","category":"reasoning","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/simplebench"},{"benchmark":"Artificial Analysis — Coding Index","benchmark_slug":"aa-coding-index","score":38.4,"unit":"index","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-coding-index"},{"benchmark":"Artificial Analysis · Terminal-Bench Hard","benchmark_slug":"aa-terminal-bench-hard","score":37.1,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-terminal-bench-hard"},{"benchmark":"Artificial Analysis — Agentic Index","benchmark_slug":"aa-agentic-index","score":36.09,"unit":"index","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-agentic-index"},{"benchmark":"Chess Puzzles","benchmark_slug":"chess-puzzles","score":34.76,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/chess-puzzles"},{"benchmark":"FrontierMath-Tiers-1-3-v2-Private","benchmark_slug":"frontiermath-tiers-1-3-v2-private","score":33.33,"unit":"%","category":"math","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/frontiermath-tiers-1-3-v2-private"},{"benchmark":"Gdpval","benchmark_slug":"gdpval","score":30.8,"unit":"%","category":"general","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/gdpval"},{"benchmark":"VPCT","benchmark_slug":"vpct","score":28,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/vpct"},{"benchmark":"OSWorld","benchmark_slug":"osworld","score":23,"unit":"%","category":"agentic","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/osworld"},{"benchmark":"Mystery Game Puzzles","benchmark_slug":"mystery-game-puzzles","score":21.79,"unit":"%","category":"general","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/mystery-game-puzzles"},{"benchmark":"Artificial Analysis — Quality Index","benchmark_slug":"aa-quality-index","score":20.2,"unit":"index","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-quality-index"},{"benchmark":"Artificial Analysis · Humanity's Last Exam","benchmark_slug":"aa-humanitys-last-exam","score":20.1,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-humanitys-last-exam"},{"benchmark":"FrontierMath-2025-02-28-Private","benchmark_slug":"frontiermath-2025-02-28-private","score":18.69,"unit":"%","category":"math","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/frontiermath-2025-02-28-private"},{"benchmark":"Cl Bench","benchmark_slug":"cl-bench","score":17.8,"unit":"%","category":"general","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/cl-bench"},{"benchmark":"APEX-Agents","benchmark_slug":"apex-agents","score":17.2,"unit":"%","category":"agentic","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/apex-agents"},{"benchmark":"HLE","benchmark_slug":"hle","score":16.3,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/hle"},{"benchmark":"EnigmaEval","benchmark_slug":"seal-enigmaeval","score":13.09,"unit":"%","category":"knowledge","source":"Scale SEAL leaderboards","source_url":"https://scale.com/leaderboard","benchmark_url":"https://benchgecko.ai/benchmark/seal-enigmaeval"},{"benchmark":"GSO-Bench","benchmark_slug":"gso-bench","score":8.8,"unit":"%","category":"coding","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/gso-bench"},{"benchmark":"ARC-AGI-2","benchmark_slug":"arc-agi-2","score":6.53,"unit":"%","category":"reasoning","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/arc-agi-2"},{"benchmark":"FrontierMath-Tier-4-2025-07-01-Private","benchmark_slug":"frontiermath-tier-4-2025-07-01-private","score":3.47,"unit":"%","category":"math","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/frontiermath-tier-4-2025-07-01-private"},{"benchmark":"Artificial Analysis · CritPt","benchmark_slug":"aa-critpt","score":1.1,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-critpt"}],"gecko_tests":null,"links":{"page":"https://benchgecko.ai/model/o3","json":"https://benchgecko.ai/api/v1/models/o3","price_history":"https://benchgecko.ai/api/v1/price-history/o3","provider_prices":null}}