{"source":"BenchGecko","url":"https://benchgecko.ai/model/claude-3-5-sonnet","as_of":"2026-03-27","license":"BenchGecko collected data (prices, provider offers, Gecko Tests) is CC BY 4.0 · https://creativecommons.org/licenses/by/4.0/. Benchmark scores belong to their original publishers (see sources) and are aggregated with attribution. Attribution required: \"Source: BenchGecko\" with a link.","attribution":"Source: BenchGecko · https://benchgecko.ai/model/claude-3-5-sonnet","cite":"Claude 3.5 Sonnet · benchmarks, pricing and providers. BenchGecko, data as of 2026-03-27. https://benchgecko.ai/model/claude-3-5-sonnet","slug":"claude-3-5-sonnet","name":"Claude 3.5 Sonnet","provider":{"name":"Anthropic","slug":"anthropic"},"model_type":"text","release_date":"2024-01-01","is_open_source":false,"status":"benchmark-only","context_window":null,"description":null,"benchgecko_score":{"value":39.8,"rank":198,"of":312,"method":"Normalized average of public benchmark scores","method_url":"https://benchgecko.ai/methodology"},"avg_score":39.8,"list_price":{"input_usd_per_m":null,"output_usd_per_m":null,"source":"OpenRouter models API (checked daily)","as_of":"2026-10-05"},"pricing":{"input":null,"output":null},"providers":[],"cheapest_provider":null,"providers_source":null,"scores":[{"benchmark":"Chatbot Arena Elo — Overall","benchmark_slug":"arena-elo-overall","score":1373.99,"unit":"elo","category":"arena","source":"LMArena","source_url":"https://lmarena.ai/leaderboard","benchmark_url":"https://benchgecko.ai/benchmark/arena-elo-overall"},{"benchmark":"HELM — IFEval","benchmark_slug":"helm-ifeval","score":85.6,"unit":"%","category":"language","source":"Stanford HELM","source_url":"https://crfm.stanford.edu/helm/","benchmark_url":"https://benchgecko.ai/benchmark/helm-ifeval"},{"benchmark":"Aider — Code Editing","benchmark_slug":"aider-edit","score":84.2,"unit":"%","category":"coding","source":"Aider leaderboards","source_url":"https://aider.chat/docs/leaderboards/","benchmark_url":"https://benchgecko.ai/benchmark/aider-edit"},{"benchmark":"MMLU","benchmark_slug":"mmlu","score":82,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/mmlu"},{"benchmark":"Lech Mazur Writing","benchmark_slug":"lech-mazur-writing","score":80.3,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/lech-mazur-writing"},{"benchmark":"HELM — WildBench","benchmark_slug":"helm-wildbench","score":79.2,"unit":"%","category":"reasoning","source":"Stanford HELM","source_url":"https://crfm.stanford.edu/helm/","benchmark_url":"https://benchgecko.ai/benchmark/helm-wildbench"},{"benchmark":"HELM — MMLU-Pro","benchmark_slug":"helm-mmlu-pro","score":77.7,"unit":"%","category":"knowledge","source":"Stanford HELM","source_url":"https://crfm.stanford.edu/helm/","benchmark_url":"https://benchgecko.ai/benchmark/helm-mmlu-pro"},{"benchmark":"GeoBench","benchmark_slug":"geobench","score":62,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/geobench"},{"benchmark":"HELM — GPQA","benchmark_slug":"helm-gpqa","score":56.5,"unit":"%","category":"knowledge","source":"Stanford HELM","source_url":"https://crfm.stanford.edu/helm/","benchmark_url":"https://benchgecko.ai/benchmark/helm-gpqa"},{"benchmark":"MATH level 5","benchmark_slug":"math-level-5","score":51.68,"unit":"%","category":"math","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/math-level-5"},{"benchmark":"Aider polyglot","benchmark_slug":"aider-polyglot","score":51.6,"unit":"%","category":"coding","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/aider-polyglot"},{"benchmark":"CadEval","benchmark_slug":"cadeval","score":48,"unit":"%","category":"coding","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/cadeval"},{"benchmark":"VideoMME","benchmark_slug":"videomme","score":46.67,"unit":"%","category":"multimodal","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/videomme"},{"benchmark":"Dtbench","benchmark_slug":"dtbench","score":46.28,"unit":"%","category":"general","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/dtbench"},{"benchmark":"Metr Time Horizons","benchmark_slug":"metr-time-horizons","score":40.15,"unit":"%","category":"general","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/metr-time-horizons"},{"benchmark":"GPQA diamond","benchmark_slug":"gpqa-diamond","score":38.72,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/gpqa-diamond"},{"benchmark":"Balrog","benchmark_slug":"balrog","score":32.6,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/balrog"},{"benchmark":"WeirdML","benchmark_slug":"weirdml","score":30.97,"unit":"%","category":"coding","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/weirdml"},{"benchmark":"HELM — Omni-MATH","benchmark_slug":"helm-omni-math","score":27.6,"unit":"%","category":"math","source":"Stanford HELM","source_url":"https://crfm.stanford.edu/helm/","benchmark_url":"https://benchgecko.ai/benchmark/helm-omni-math"},{"benchmark":"The Agent Company","benchmark_slug":"the-agent-company","score":24,"unit":"%","category":"agentic","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/the-agent-company"},{"benchmark":"Cybench","benchmark_slug":"cybench","score":17.5,"unit":"%","category":"coding","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/cybench"},{"benchmark":"SimpleBench","benchmark_slug":"simplebench","score":13,"unit":"%","category":"reasoning","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/simplebench"},{"benchmark":"Fortress","benchmark_slug":"seal-fortress","score":12.96,"unit":"%","category":"safety","source":"Scale SEAL leaderboards","source_url":"https://scale.com/leaderboard","benchmark_url":"https://benchgecko.ai/benchmark/seal-fortress"},{"benchmark":"OTIS Mock AIME 2024-2025","benchmark_slug":"otis-mock-aime-2024-2025","score":6.43,"unit":"%","category":"math","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/otis-mock-aime-2024-2025"},{"benchmark":"GSO-Bench","benchmark_slug":"gso-bench","score":4.6,"unit":"%","category":"coding","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/gso-bench"},{"benchmark":"FrontierMath-2025-02-28-Private","benchmark_slug":"frontiermath-2025-02-28-private","score":1.81,"unit":"%","category":"math","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/frontiermath-2025-02-28-private"},{"benchmark":"FrontierMath-Tier-4-2025-07-01-Private","benchmark_slug":"frontiermath-tier-4-2025-07-01-private","score":0.1,"unit":"%","category":"math","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/frontiermath-tier-4-2025-07-01-private"},{"benchmark":"HLE","benchmark_slug":"hle","score":0,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/hle"},{"benchmark":"VPCT","benchmark_slug":"vpct","score":0,"unit":"%","category":"knowledge","source":"Epoch AI Benchmarking Hub","source_url":"https://epoch.ai/data/ai-benchmarking-dashboard","benchmark_url":"https://benchgecko.ai/benchmark/vpct"}],"gecko_tests":null,"links":{"page":"https://benchgecko.ai/model/claude-3-5-sonnet","json":"https://benchgecko.ai/api/v1/models/claude-3-5-sonnet","price_history":"https://benchgecko.ai/api/v1/price-history/claude-3-5-sonnet","provider_prices":null}}