{"source":"BenchGecko","url":"https://benchgecko.ai/model/phi-4","as_of":"2026-10-05","license":"BenchGecko collected data (prices, provider offers, Gecko Tests) is CC BY 4.0 · https://creativecommons.org/licenses/by/4.0/. Benchmark scores belong to their original publishers (see sources) and are aggregated with attribution. Attribution required: \"Source: BenchGecko\" with a link.","attribution":"Source: BenchGecko · https://benchgecko.ai/model/phi-4","cite":"Phi 4 · benchmarks, pricing and providers. BenchGecko, data as of 2026-10-05. https://benchgecko.ai/model/phi-4","slug":"phi-4","name":"Phi 4","provider":{"name":"Microsoft","slug":"microsoft"},"model_type":"text","release_date":"2025-01-10","is_open_source":true,"status":"active","context_window":16384,"description":"Microsoft Research Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","benchgecko_score":{"value":49,"rank":151,"of":312,"method":"Normalized average of public benchmark scores","method_url":"https://benchgecko.ai/methodology"},"avg_score":49,"list_price":{"input_usd_per_m":0.07,"output_usd_per_m":0.14,"source":"OpenRouter models API (checked daily)","as_of":"2026-10-05"},"pricing":{"input":0.07,"output":0.14},"providers":[{"provider":"DeepInfra","endpoint":"deepinfra/bf16","quantization":"bf16","context_tokens":16384,"input_usd_per_m":0.07,"output_usd_per_m":0.14,"cache_read_usd_per_m":null,"uptime_1d_pct":100,"as_of":"2026-10-05"}],"cheapest_provider":{"provider":"DeepInfra","endpoint":"deepinfra/bf16","quantization":"bf16","context_tokens":16384,"input_usd_per_m":0.07,"output_usd_per_m":0.14,"cache_read_usd_per_m":null,"uptime_1d_pct":100,"as_of":"2026-10-05"},"providers_source":"OpenRouter endpoints API, one row per provider, refreshed daily","scores":[{"benchmark":"Chatbot Arena Elo — Overall","benchmark_slug":"arena-elo-overall","score":1256.27,"unit":"elo","category":"arena","source":"LMArena","source_url":"https://lmarena.ai/leaderboard","benchmark_url":"https://benchgecko.ai/benchmark/arena-elo-overall"},{"benchmark":"MMLU","benchmark_slug":"mmlu","score":79.73,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/mmlu"},{"benchmark":"MATH level 5","benchmark_slug":"math-level-5","score":64.94,"unit":"%","category":"math","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/math-level-5"},{"benchmark":"IFEval","benchmark_slug":"hf-ifeval","score":64.84,"unit":"%","category":"language","source":"Hugging Face Open LLM Leaderboard (archived)","source_url":"https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard","benchmark_url":"https://benchgecko.ai/benchmark/hf-ifeval"},{"benchmark":"Lech Mazur Writing","benchmark_slug":"lech-mazur-writing","score":62.6,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/lech-mazur-writing"},{"benchmark":"Artificial Analysis · GPQA Diamond","benchmark_slug":"aa-gpqa-diamond","score":57.5,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-gpqa-diamond"},{"benchmark":"BBH (HuggingFace)","benchmark_slug":"hf-bbh","score":55.67,"unit":"%","category":"general","source":"Hugging Face Open LLM Leaderboard (archived)","source_url":"https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard","benchmark_url":"https://benchgecko.ai/benchmark/hf-bbh"},{"benchmark":"MMLU-PRO","benchmark_slug":"hf-mmlu-pro","score":48.34,"unit":"%","category":"knowledge","source":"Hugging Face Open LLM Leaderboard (archived)","source_url":"https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard","benchmark_url":"https://benchgecko.ai/benchmark/hf-mmlu-pro"},{"benchmark":"MATH Level 5","benchmark_slug":"hf-math-lvl5","score":45.24,"unit":"%","category":"math","source":"Hugging Face Open LLM Leaderboard (archived)","source_url":"https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard","benchmark_url":"https://benchgecko.ai/benchmark/hf-math-lvl5"},{"benchmark":"GPQA diamond","benchmark_slug":"gpqa-diamond","score":41.41,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/gpqa-diamond"},{"benchmark":"Artificial Analysis · IFBench","benchmark_slug":"aa-ifbench","score":23.5,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-ifbench"},{"benchmark":"OTIS Mock AIME 2024-2025","benchmark_slug":"otis-mock-aime-2024-2025","score":13.66,"unit":"%","category":"math","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/otis-mock-aime-2024-2025"},{"benchmark":"Balrog","benchmark_slug":"balrog","score":11.6,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/balrog"},{"benchmark":"MUSR","benchmark_slug":"hf-musr","score":11.43,"unit":"%","category":"reasoning","source":"Hugging Face Open LLM Leaderboard (archived)","source_url":"https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard","benchmark_url":"https://benchgecko.ai/benchmark/hf-musr"},{"benchmark":"Artificial Analysis — Coding Index","benchmark_slug":"aa-coding-index","score":11.21,"unit":"index","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-coding-index"},{"benchmark":"GPQA","benchmark_slug":"hf-gpqa","score":9.17,"unit":"%","category":"knowledge","source":"Hugging Face Open LLM Leaderboard (archived)","source_url":"https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard","benchmark_url":"https://benchgecko.ai/benchmark/hf-gpqa"},{"benchmark":"Artificial Analysis — Quality Index","benchmark_slug":"aa-quality-index","score":5.92,"unit":"index","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-quality-index"},{"benchmark":"Artificial Analysis · Terminal-Bench Hard","benchmark_slug":"aa-terminal-bench-hard","score":3.8,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-terminal-bench-hard"},{"benchmark":"Artificial Analysis · Humanity's Last Exam","benchmark_slug":"aa-humanitys-last-exam","score":3.8,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-humanitys-last-exam"},{"benchmark":"Chess Puzzles","benchmark_slug":"chess-puzzles","score":0,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/chess-puzzles"},{"benchmark":"Artificial Analysis — Agentic Index","benchmark_slug":"aa-agentic-index","score":0,"unit":"index","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-agentic-index"},{"benchmark":"Artificial Analysis · Long Context Reasoning","benchmark_slug":"aa-long-context-reasoning","score":0,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-long-context-reasoning"},{"benchmark":"Artificial Analysis · CritPt","benchmark_slug":"aa-critpt","score":0,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-critpt"},{"benchmark":"Artificial Analysis · tau2-Bench Telecom","benchmark_slug":"aa-tau2-bench","score":0,"unit":"%","category":"speed","source":"Artificial Analysis","source_url":"https://artificialanalysis.ai","benchmark_url":"https://benchgecko.ai/benchmark/aa-tau2-bench"}],"gecko_tests":null,"links":{"page":"https://benchgecko.ai/model/phi-4","json":"https://benchgecko.ai/api/v1/models/phi-4","price_history":"https://benchgecko.ai/api/v1/price-history/phi-4","provider_prices":null}}