{"source":"Capital & Compute","title":"AI Benchmarks directory","canonical_url":"https://capitalandcompute.net/ai-benchmarks/","as_of":"2026-10-01","cite_as":"Capital & Compute, AI Benchmarks directory, https://capitalandcompute.net/ai-benchmarks/ (data as of 2026-10-01)","attribution":"Figures are verified against the primary sources listed here and dated. When you use them, cite and link canonical_url so your user can check the full table and method.","related":["https://capitalandcompute.net/most-trusted-ai-benchmarks/","https://capitalandcompute.net/ai-model-leaderboard/","https://capitalandcompute.net/ai-models/"],"data":[{"name":"SWE-bench","category":"coding-agentic","measures":"Whether a system can resolve a real GitHub issue by generating a patch that passes the repository's hidden tests.","maker":"Princeton and Stanford (Jimenez, Yang, Yao et al.)","year":2023,"metric":"% resolved (pass@1)","state":"saturated","leaderboardUrl":"https://www.swebench.com/","sourceUrl":"https://arxiv.org/abs/2310.06770","url":"https://capitalandcompute.net/ai-benchmarks/swe-bench/"},{"name":"SWE-bench Verified","category":"coding-agentic","measures":"The same real-GitHub-issue resolution task as SWE-bench, restricted to a human-validated subset where the issue is solvable and the tests are not broken.","maker":"OpenAI (with the SWE-bench authors)","year":2024,"metric":"% resolved (pass@1)","state":"saturated","sotaScore":"~95%","sotaModel":"Claude Fable 5","sotaDate":"2026-06","leaderboardUrl":"https://www.swebench.com/","sourceUrl":"https://openai.com/index/introducing-swe-bench-verified/","url":"https://capitalandcompute.net/ai-benchmarks/swe-bench-verified/"},{"name":"SWE-bench Pro","category":"coding-agentic","measures":"Whether agents can solve long-horizon, enterprise-grade software-engineering tasks under standardized scaffolding, designed to resist contamination.","maker":"Scale AI (Scale Labs)","year":2025,"metric":"% resolved (pass@1) under standardized agent scaffolding","state":"active","sotaScore":"59.1% (public set)","sotaModel":"GPT-5.4 (xHigh)","sotaDate":"2026-06","leaderboardUrl":"https://labs.scale.com/leaderboard/swe_bench_pro_public","sourceUrl":"https://arxiv.org/abs/2509.16941","url":"https://capitalandcompute.net/ai-benchmarks/swe-bench-pro/"},{"name":"SWE-bench Multimodal","category":"coding-agentic","measures":"Whether coding agents can resolve real GitHub issues in visual, user-facing JavaScript software where the bug or feature involves the UI.","maker":"Stanford and Princeton (Yang, Jimenez et al.)","year":2024,"metric":"% resolved (pass@1)","state":"active","leaderboardUrl":"https://www.swebench.com/multimodal.html","sourceUrl":"https://arxiv.org/abs/2410.03859"},{"name":"DeepSWE","category":"coding-agentic","measures":"Whether frontier coding agents can complete original, long-horizon engineering tasks written from scratch, with no upstream PR to memorize.","maker":"Datacurve","year":2026,"metric":"pass@1 (committed code graded in a clean environment)","state":"active","sotaScore":"74% (v1.1)","sotaModel":"GPT-6 Astra, Gemini 3.8 Flash and Claude Opus 5 (three-way tie)","sotaDate":"2026-09","leaderboardUrl":"https://deepswe.datacurve.ai/","sourceUrl":"https://github.com/datacurve-ai/deep-swe"},{"name":"FrontierCode","category":"coding-agentic","measures":"Whether a coding agent produces a mergeable, production-quality pull request, not just one that passes tests, judged on correctness, regression safety, scope, tests and style.","maker":"Cognition (with 20+ open-source maintainers)","year":2026,"metric":"Pass rate on blocker criteria plus a weighted six-dimension quality rubric","state":"active","sotaScore":"13.4% (Diamond)","sotaModel":"Claude Opus 4.8","sotaDate":"2026-06","sourceUrl":"https://cognition.com/blog/frontier-code"},{"name":"Terminal-Bench","category":"coding-agentic","measures":"Whether an AI agent can complete hard, realistic command-line tasks (build, configure, train, debug, secure) end to end inside a real terminal.","maker":"The Terminal-Bench team (Marten, Shaw, Merrill) with the Laude Institute and 100+ community task contributors","year":2026,"metric":"Pass/fail per trial, graded by a verification script in a separate verifier container; each task is attempted 5 times, so a full board run is 330 trials","state":"active","sotaScore":"58.2% (4.0)","sotaModel":"GPT-6 Astra (Codex, max effort)","sotaDate":"2026-09","leaderboardUrl":"https://www.tbench.ai/leaderboard/terminal-bench/4.0","sourceUrl":"https://arxiv.org/abs/2601.11868","url":"https://capitalandcompute.net/ai-benchmarks/terminal-bench/"},{"name":"SWE-Lancer","category":"coding-agentic","measures":"Whether frontier models can complete real paid freelance software jobs, both coding and technical-management tasks, well enough to earn the payouts.","maker":"OpenAI (Miserendino, Patwardhan et al.)","year":2025,"metric":"Dollars earned (and % of tasks resolved)","state":"active","sourceUrl":"https://arxiv.org/abs/2502.12115","url":"https://capitalandcompute.net/ai-benchmarks/swe-lancer/"},{"name":"Aider Polyglot","category":"coding-agentic","measures":"How well a model writes and correctly edits code across many languages, including applying diffs in the right format and self-correcting after test failures.","maker":"Aider (Paul Gauthier)","year":2024,"metric":"Percent correct after the second attempt, plus percent using the correct edit format","state":"active","leaderboardUrl":"https://aider.chat/docs/leaderboards/","sourceUrl":"https://aider.chat/2024/12/21/polyglot.html","url":"https://capitalandcompute.net/ai-benchmarks/aider-polyglot/"},{"name":"LiveCodeBench","category":"coding-agentic","measures":"Code generation and related skills (self-repair, execution, test-output prediction) on fresh competitive-programming problems, designed to be contamination-free.","maker":"UC Berkeley, MIT and Cornell (Jain, Han et al.)","year":2024,"metric":"pass@1","state":"active","leaderboardUrl":"https://livecodebench.github.io/leaderboard.html","sourceUrl":"https://arxiv.org/abs/2403.07974","url":"https://capitalandcompute.net/ai-benchmarks/livecodebench/"},{"name":"BigCodeBench","category":"coding-agentic","measures":"Whether models can write code that correctly invokes multiple function calls from diverse real libraries to satisfy complex, practical instructions.","maker":"BigCode project (Zhuo et al.)","year":2024,"metric":"pass@1 against rigorous per-task test suites","state":"active","leaderboardUrl":"https://huggingface.co/spaces/bigcode/bigcodebench-leaderboard","sourceUrl":"https://arxiv.org/abs/2406.15877"},{"name":"RepoBench","category":"coding-agentic","measures":"Repository-level code auto-completion: retrieving relevant cross-file context, predicting the next line, and the combined retrieval-plus-completion pipeline.","maker":"Liu, Xu and McAuley (UC San Diego)","year":2023,"metric":"Retrieval accuracy and exact-match / edit similarity for next-line completion","state":"active","sourceUrl":"https://arxiv.org/abs/2306.03091"},{"name":"Multi-SWE-bench","category":"coding-agentic","measures":"Cross-language issue resolution: whether agents can resolve real GitHub issues with a passing patch across many languages beyond Python.","maker":"ByteDance (ByteDance Seed)","year":2025,"metric":"% resolved (pass@1)","state":"active","leaderboardUrl":"https://multi-swe-bench.github.io/","sourceUrl":"https://arxiv.org/abs/2504.02605"},{"name":"HumanEval","category":"coding-agentic","measures":"Whether a model can synthesize a single correct Python function from a docstring so that it passes the provided unit tests.","maker":"OpenAI (Chen et al.)","year":2021,"metric":"pass@k (primarily pass@1)","state":"saturated","sotaScore":"~99%","sotaModel":"Frontier models broadly","sotaDate":"2025-04","sourceUrl":"https://arxiv.org/abs/2107.03374","url":"https://capitalandcompute.net/ai-benchmarks/humaneval/"},{"name":"MBPP","category":"coding-agentic","measures":"Whether a model can generate short, entry-level Python functions from a natural-language prompt that pass the provided tests.","maker":"Google Research (Austin, Odena et al.)","year":2021,"metric":"pass@1","state":"saturated","sotaScore":"~95%+","sotaModel":"Frontier models broadly","sotaDate":"2026-06","sourceUrl":"https://arxiv.org/abs/2108.07732"},{"name":"Frontier-Bench (now Terminal-Bench 3.0)","category":"coding-agentic","measures":"Whether a coding agent can do senior-level engineering work: building features from realistic instructions, investigating bugs that need runtime inspection, and shipping code that matches an existing repository's conventions.","maker":"The Terminal-Bench and Harbor team (Marten, Shaw, Konwinski) with 100+ task contributors and reviewers","year":2026,"metric":"Resolution rate (mean reward over repeated attempts), reported alongside cost and token use","state":"retired","sotaScore":"34.4%","sotaModel":"GPT-5.6 Sol","sotaDate":"2026-07","leaderboardUrl":"https://www.tbench.ai/leaderboard/terminal-bench/4.0","sourceUrl":"https://www.tbench.ai/news/terminal-bench-3-0"},{"name":"CursorBench","category":"coding-agentic","measures":"Whether a coding agent can handle ambiguous, multi-file requests inside a real repository, judged on solution correctness, code quality, efficiency and interaction behaviour.","maker":"Anysphere (Cursor)","year":2026,"metric":"Agentic graders scoring correctness plus quality dimensions, since the requests are underspecified and admit several valid solutions","state":"active","leaderboardUrl":"https://cursor.com/cursorbench","sourceUrl":"https://cursor.com/blog/cursorbench"},{"name":"GAIA","category":"agentic-tooluse","measures":"Whether an AI assistant can answer real-world questions that require multi-step reasoning, multiple modalities, web browsing and general tool use.","maker":"Meta AI and Hugging Face (Mialon, Fourrier et al.)","year":2023,"metric":"Exact-match accuracy against an unambiguous answer","state":"active","sotaScore":"~75%","sotaModel":"HAL agent (Claude Sonnet 4.5)","sotaDate":"2026-06","leaderboardUrl":"https://hal.cs.princeton.edu/gaia","sourceUrl":"https://arxiv.org/abs/2311.12983","url":"https://capitalandcompute.net/ai-benchmarks/gaia/"},{"name":"tau-bench","category":"agentic-tooluse","measures":"Whether a tool-using agent can reliably complete customer-service tasks over multi-turn conversations with a simulated user while obeying domain policies.","maker":"Sierra (Yao, Shinn, Narasimhan et al.)","year":2024,"metric":"pass^k: the probability an agent succeeds across all k independent trials (reliability, not just average success)","state":"active","sourceUrl":"https://arxiv.org/abs/2406.12045","url":"https://capitalandcompute.net/ai-benchmarks/tau-bench/"},{"name":"AgentBench","category":"agentic-tooluse","measures":"How well an LLM acts as an autonomous agent in multi-turn, open-ended decision-making across diverse interactive environments.","maker":"Tsinghua University (THUDM; Liu et al.)","year":2023,"metric":"Per-environment success aggregated into an overall score","state":"active","leaderboardUrl":"https://llmbench.ai/agent","sourceUrl":"https://arxiv.org/abs/2308.03688"},{"name":"WebArena","category":"agentic-tooluse","measures":"Whether an autonomous agent can complete long-horizon, realistic web tasks (navigation, forms, multi-step workflows) in fully functional self-hosted websites.","maker":"Carnegie Mellon University (Zhou, Xu et al.)","year":2023,"metric":"Functional success rate via execution-based reward checking the end state","state":"active","sotaScore":"74.3% (third-party tracker)","sotaModel":"WebTactix on DeepSeek v3.2 (a system, not a bare model)","sotaDate":"2026-06","leaderboardUrl":"https://leaderboard.steel.dev/leaderboards/webarena/","sourceUrl":"https://arxiv.org/abs/2307.13854","url":"https://capitalandcompute.net/ai-benchmarks/webarena/"},{"name":"VisualWebArena","category":"agentic-tooluse","measures":"Whether a multimodal agent can complete visually grounded web tasks that require interpreting images and page layout, not just text.","maker":"Carnegie Mellon University (Koh et al.)","year":2024,"metric":"Functional success rate via execution-based evaluation","state":"active","leaderboardUrl":"https://jykoh.com/vwa","sourceUrl":"https://arxiv.org/abs/2401.13649"},{"name":"OSWorld","category":"agentic-tooluse","measures":"Whether a multimodal agent can operate a real computer (desktop apps, file I/O, multi-app workflows) to complete open-ended tasks in a live virtual machine.","maker":"XLANG Lab, University of Hong Kong (Xie et al.)","year":2024,"metric":"Execution-based success rate via per-task verification scripts that inspect machine state","state":"saturated","leaderboardUrl":"https://os-world.github.io/","sourceUrl":"https://arxiv.org/abs/2404.07972"},{"name":"BrowseComp","category":"agentic-tooluse","measures":"Whether a browsing agent can persistently navigate the open web to locate a single hard-to-find, entangled fact.","maker":"OpenAI (Wei, Sun et al.)","year":2025,"metric":"Accuracy via model-graded semantic equivalence to the reference answer","state":"active","sotaScore":"51.5%","sotaModel":"OpenAI Deep Research (launch paper)","sotaDate":"2025-04","sourceUrl":"https://arxiv.org/abs/2504.12516"},{"name":"MLE-bench","category":"agentic-tooluse","measures":"Whether an AI agent can do end-to-end machine-learning engineering (data prep, training, experimentation, submission) at the level of human Kaggle competitors.","maker":"OpenAI (Chan et al.)","year":2024,"metric":"Medal rate (fraction of competitions reaching bronze/silver/gold thresholds)","state":"active","sotaScore":"16.9% (paper baseline)","sotaModel":"o1-preview with AIDE scaffolding","sotaDate":"2024-10","leaderboardUrl":"https://github.com/openai/mle-bench","sourceUrl":"https://arxiv.org/abs/2410.07095"},{"name":"OSWorld 2.0","category":"agentic-tooluse","measures":"Whether a computer-use agent can finish long-horizon professional work on a real desktop, coordinating several applications and leaving the machine in the correct final state.","maker":"XLANG Lab, University of Hong Kong, with collaborators at Columbia, UCSB, UCSD, Mila, Ohio State and others","year":2026,"metric":"Binary completion at a 500-step cap, reported with a weighted-checkpoint partial score","state":"active","sotaScore":"20.6%","sotaModel":"Claude Opus 4.8 (max thinking)","sotaDate":"2026-06","leaderboardUrl":"https://osworld-v2.xlang.ai/","sourceUrl":"https://arxiv.org/abs/2606.29537","url":"https://capitalandcompute.net/ai-benchmarks/osworld-2/"},{"name":"AutomationBench","category":"agentic-tooluse","measures":"Whether an agent can run a realistic business workflow end to end across several apps: discovering the right API endpoints itself, following a policy document, and writing correct data into every system it touches.","maker":"Zapier (Shepard and Salimans)","year":2026,"metric":"task_completed_correctly: strict pass/fail where every scored end-state assertion must pass, with partial credit reported only as a diagnostic","state":"active","sotaScore":"51.3% (v1.0.6)","sotaModel":"Gemini 4 Argon (high effort)","sotaDate":"2026-09","leaderboardUrl":"https://zapier.com/benchmarks","sourceUrl":"https://arxiv.org/abs/2604.18934"},{"name":"DeepSearchQA","category":"agentic-tooluse","measures":"Whether a deep-research agent can plan and execute a long chain of web searches to return an exhaustive, de-duplicated answer list rather than a single fact.","maker":"Google DeepMind (Gupta, Chatterjee, Haas et al.)","year":2026,"metric":"Accuracy against each task's objectively verifiable exhaustive answer set","state":"active","leaderboardUrl":"https://www.kaggle.com/benchmarks/google/dsqa","sourceUrl":"https://arxiv.org/abs/2601.20975"},{"name":"ARC-AGI-1","category":"reasoning","measures":"Whether a system can infer the abstract rule of a novel visual grid puzzle from a few examples and apply it to a new input.","maker":"Francois Chollet (ARC Prize Foundation)","year":2019,"metric":"pass@2 exact-grid-match accuracy","state":"saturated","sotaScore":"97.5% (public eval)","sotaModel":"Claude Opus 5 and GPT-5.6 Sol","sotaDate":"2026-07","leaderboardUrl":"https://arcprize.org/leaderboard","sourceUrl":"https://arcprize.org/arc-agi/1","url":"https://capitalandcompute.net/ai-benchmarks/arc-agi-1/"},{"name":"ARC-AGI-2","category":"reasoning","measures":"The same fluid-intelligence test as ARC-AGI-1, but with harder, contamination-resistant tasks that stay easy for humans yet very hard for AI.","maker":"ARC Prize Foundation (Chollet et al.)","year":2025,"metric":"pass@2 exact-grid-match accuracy, reported with a cost-per-task efficiency metric","state":"saturated","sotaScore":"92.5% (semi-private)","sotaModel":"GPT-5.6 Sol (max effort)","sotaDate":"2026-07","leaderboardUrl":"https://arcprize.org/leaderboard","sourceUrl":"https://arcprize.org/arc-agi/2","url":"https://capitalandcompute.net/ai-benchmarks/arc-agi-2/"},{"name":"GPQA Diamond","category":"reasoning","measures":"Graduate and PhD-level multiple-choice scientific reasoning in biology, physics and chemistry, on questions designed to be unanswerable by quick web search.","maker":"Rein et al. (NYU, Cohere, Anthropic)","year":2023,"metric":"Multiple-choice accuracy (random baseline 25%, PhD-expert baseline about 70%)","state":"saturated","sotaScore":"~94%","sotaModel":"Gemini 3.1 Pro Preview","sotaDate":"2026-02","leaderboardUrl":"https://epoch.ai/benchmarks/gpqa-diamond","sourceUrl":"https://arxiv.org/abs/2311.12022","url":"https://capitalandcompute.net/ai-benchmarks/gpqa-diamond/"},{"name":"Humanity's Last Exam","category":"reasoning","measures":"Frontier, closed-ended expert knowledge and reasoning across more than 100 academic disciplines at the limit of human expertise.","maker":"Center for AI Safety (CAIS) and Scale AI","year":2025,"metric":"Accuracy (exact match / multiple-choice), often reported with a calibration metric","state":"active","sotaScore":"53.3%","sotaModel":"Claude Fable 5 (Max Effort)","sotaDate":"2026-06","leaderboardUrl":"https://artificialanalysis.ai/evaluations/humanitys-last-exam","sourceUrl":"https://arxiv.org/abs/2501.14249","url":"https://capitalandcompute.net/ai-benchmarks/humanitys-last-exam/"},{"name":"BIG-Bench Hard","category":"reasoning","measures":"A suite of multi-step reasoning tasks (logic, arithmetic, algorithmic, commonsense) on which pre-2022 models trailed average human raters.","maker":"Suzgun et al. (Google Research and Stanford)","year":2022,"metric":"Per-task accuracy averaged across the 23 tasks","state":"saturated","sourceUrl":"https://github.com/suzgunmirac/BIG-Bench-Hard"},{"name":"MuSR","category":"reasoning","measures":"Multistep commonsense reasoning embedded in long natural-language narratives such as murder mysteries, object placement and team allocation.","maker":"Sprague, Ye, Durrett et al. (UT Austin)","year":2023,"metric":"Multiple-choice accuracy","state":"active","leaderboardUrl":"https://llm-stats.com/benchmarks/musr","sourceUrl":"https://arxiv.org/abs/2310.16049"},{"name":"ARC-AGI-3","category":"reasoning","measures":"Whether an agent dropped into an unfamiliar interactive environment with no instructions, stated goal or rules can work out what to do by acting, build a usable world model, and keep learning across levels.","maker":"ARC Prize Foundation","year":2026,"metric":"Games beaten at or above human-level action efficiency, measuring skill-acquisition efficiency rather than one-shot accuracy","state":"active","sotaScore":"62.7% (semi-private, standard harness)","sotaModel":"GPT-6 Astra (max effort)","sotaDate":"2026-09","leaderboardUrl":"https://arcprize.org/leaderboard","sourceUrl":"https://arxiv.org/abs/2603.24621","url":"https://capitalandcompute.net/ai-benchmarks/arc-agi-3/"},{"name":"FrontierMath","category":"math","measures":"Research-level original mathematics requiring hours to days of expert effort, across number theory, analysis, algebraic geometry and more.","maker":"Epoch AI","year":2024,"metric":"Accuracy (fraction with a correct, automatically verifiable final answer)","state":"saturated","sotaScore":"87% (Tiers 1-3)","sotaModel":"Claude Fable 5","sotaDate":"2026-06","leaderboardUrl":"https://epoch.ai/benchmarks/frontiermath","sourceUrl":"https://epoch.ai/frontiermath","url":"https://capitalandcompute.net/ai-benchmarks/frontiermath/"},{"name":"AIME 2025","category":"math","measures":"Olympiad-track competition mathematics at the level of the American Invitational Mathematics Examination, used as a high-difficulty LLM eval.","maker":"Mathematical Association of America; adopted as an LLM eval by the community","year":2025,"metric":"Exact-match accuracy, usually pass@1 averaged over samples","state":"saturated","sotaScore":"100%","sotaModel":"Multiple frontier reasoning models","sotaDate":"2026-06","leaderboardUrl":"https://matharena.ai/","sourceUrl":"https://matharena.ai/"},{"name":"MATH","category":"math","measures":"Step-by-step solving of high-school competition mathematics across algebra, geometry, number theory, probability and precalculus.","maker":"Hendrycks et al. (UC Berkeley)","year":2021,"metric":"Exact-match accuracy on the final boxed answer","state":"saturated","sotaScore":"~99% (MATH-500)","sotaModel":"GPT-5","sotaDate":"2026-04","leaderboardUrl":"https://llm-stats.com/benchmarks/math-500","sourceUrl":"https://arxiv.org/abs/2103.03874"},{"name":"GSM8K","category":"math","measures":"Multi-step grade-school arithmetic word-problem reasoning.","maker":"OpenAI (Cobbe et al.)","year":2021,"metric":"Exact-match accuracy on the final numeric answer","state":"saturated","sotaScore":"~99.6%","sotaModel":"Frontier models broadly","sotaDate":"2026-05","leaderboardUrl":"https://llm-stats.com/benchmarks/gsm8k","sourceUrl":"https://arxiv.org/abs/2110.14168","url":"https://capitalandcompute.net/ai-benchmarks/gsm8k/"},{"name":"Omni-MATH","category":"math","measures":"Olympiad-level mathematical reasoning across a broad range of subdomains and difficulty levels.","maker":"Gao, Song, Cai et al. (Peking University and collaborators)","year":2024,"metric":"Accuracy, scored with an LLM-based verifier (Omni-Judge)","state":"active","leaderboardUrl":"https://omni-math.github.io/","sourceUrl":"https://arxiv.org/abs/2410.07985"},{"name":"MathArena","category":"math","measures":"Mathematical reasoning and proof-writing on freshly released competition problems, evaluated before they can enter training data.","maker":"ETH Zurich (SRI Lab)","year":2025,"metric":"Per-competition accuracy and an aggregate expected-performance score","state":"active","sotaScore":"81.1% (aggregate)","sotaModel":"GPT-5.5 (xhigh)","sotaDate":"2026-04","leaderboardUrl":"https://matharena.ai/","sourceUrl":"https://arxiv.org/abs/2505.23281"},{"name":"MMLU","category":"knowledge","measures":"Broad academic and professional knowledge across 57 subjects via four-choice multiple-choice questions.","maker":"Hendrycks et al. (UC Berkeley and collaborators)","year":2021,"metric":"Accuracy","state":"saturated","sotaScore":"~93%","sotaModel":"Qwen3.7 Max","sotaDate":"2026-06","leaderboardUrl":"https://llm-stats.com/benchmarks/mmlu","sourceUrl":"https://arxiv.org/abs/2009.03300","url":"https://capitalandcompute.net/ai-benchmarks/mmlu/"},{"name":"MMLU-Pro","category":"knowledge","measures":"Harder multi-task reasoning and knowledge designed to de-saturate MMLU and reward deliberate reasoning over recall.","maker":"TIGER-Lab (Wang et al., University of Waterloo)","year":2024,"metric":"Accuracy","state":"active","sotaScore":"~90%","sotaModel":"Gemini 3 Pro Preview","sotaDate":"2026-06","leaderboardUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","sourceUrl":"https://arxiv.org/abs/2406.01574","url":"https://capitalandcompute.net/ai-benchmarks/mmlu-pro/"},{"name":"MMLU-Redux","category":"knowledge","measures":"A re-annotated, error-corrected subset of MMLU used to measure true knowledge accuracy without the original's label noise.","maker":"Gema et al. (University of Edinburgh and collaborators)","year":2024,"metric":"Accuracy on cleaned labels","state":"active","leaderboardUrl":"https://llm-stats.com/benchmarks/mmlu-redux","sourceUrl":"https://arxiv.org/abs/2406.04127"},{"name":"SimpleQA","category":"knowledge","measures":"Short-form parametric factuality: whether a model answers single-answer fact-seeking questions correctly and abstains when unsure.","maker":"OpenAI (Wei, Karina et al.)","year":2024,"metric":"Accuracy, plus correct-given-attempted and an F-score balancing attempts against accuracy","state":"active","leaderboardUrl":"https://llm-stats.com/benchmarks/simpleqa","sourceUrl":"https://arxiv.org/abs/2411.04368"},{"name":"RULER","category":"long-context","measures":"The real effective context length of a model by testing retrieval, multi-hop tracing, aggregation and QA at increasing sequence lengths.","maker":"NVIDIA (Hsieh, Sun et al.)","year":2024,"metric":"Weighted-average accuracy across tasks and lengths; effective length is the longest length still above threshold","state":"active","leaderboardUrl":"https://github.com/NVIDIA/RULER","sourceUrl":"https://arxiv.org/abs/2404.06654","url":"https://capitalandcompute.net/ai-benchmarks/ruler/"},{"name":"MRCR","category":"long-context","measures":"Whether a model can distinguish and retrieve the correct one among multiple near-identical requests buried in a long multi-turn conversation.","maker":"Google DeepMind (Michelangelo); open-source variant by OpenAI","year":2024,"metric":"Similarity of the model’s output to the target instance, gated by a required answer-prefix","state":"active","sourceUrl":"https://arxiv.org/abs/2409.12640"},{"name":"NoLiMa","category":"long-context","measures":"Long-context retrieval and reasoning when the question and the target fact share minimal literal word overlap, forcing latent association rather than keyword matching.","maker":"Adobe Research and LMU Munich (Modarressi et al.)","year":2025,"metric":"Accuracy at each length, relative to the model's short-context baseline","state":"active","leaderboardUrl":"https://github.com/adobe-research/NoLiMa","sourceUrl":"https://arxiv.org/abs/2502.05167"},{"name":"Needle-in-a-Haystack","category":"long-context","measures":"Whether a model can recall a single planted fact (the needle) inserted at varying depths within a long context (the haystack).","maker":"Greg Kamradt (independent)","year":2023,"metric":"Retrieval accuracy at each depth and length cell","state":"saturated","sourceUrl":"https://github.com/gkamradt/LLMTest_NeedleInAHaystack"},{"name":"LongBench","category":"long-context","measures":"Comprehensive long-context understanding across realistic tasks (QA, summarization, few-shot, code, synthetic) in English and Chinese.","maker":"Tsinghua University (THUDM; Bai et al.)","year":2023,"metric":"v1: per-task automatic metrics. v2: multiple-choice accuracy","state":"active","sotaScore":"57.7% (v2, with reasoning)","sotaModel":"o1-preview","sotaDate":"2024-12","leaderboardUrl":"https://longbench2.github.io/","sourceUrl":"https://arxiv.org/abs/2412.15204"},{"name":"MMMU","category":"multimodal","measures":"College-level multimodal understanding and reasoning over images, diagrams, charts and text across many disciplines.","maker":"MMMU team (Yue et al.)","year":2023,"metric":"Accuracy","state":"active","sotaScore":"~86%","sotaModel":"Qwen3.6 Plus","sotaDate":"2026-06","leaderboardUrl":"https://mmmu-benchmark.github.io/","sourceUrl":"https://arxiv.org/abs/2311.16502","url":"https://capitalandcompute.net/ai-benchmarks/mmmu/"},{"name":"MMMU-Pro","category":"multimodal","measures":"A harder, contamination-resistant version of MMMU that forces genuine visual reasoning rather than text-only shortcuts.","maker":"MMMU team (Yue et al.)","year":2024,"metric":"Accuracy","state":"active","sotaScore":"~84%","sotaModel":"Gemini 3.5 Flash","sotaDate":"2026-06","leaderboardUrl":"https://mmmu-benchmark.github.io/","sourceUrl":"https://arxiv.org/abs/2409.02813"},{"name":"MathVista","category":"multimodal","measures":"Mathematical and quantitative reasoning grounded in visual contexts such as figures, charts, geometry and scientific diagrams.","maker":"Lu et al. (UCLA, University of Washington, Microsoft Research)","year":2023,"metric":"Accuracy","state":"active","sotaScore":"~91% (testmini)","sotaModel":"Seed 2.1 Pro","sotaDate":"2026-06","leaderboardUrl":"https://mathvista.github.io/","sourceUrl":"https://arxiv.org/abs/2310.02255"},{"name":"Video-MME","category":"multimodal","measures":"Comprehensive video understanding by multimodal LLMs across short, medium and long clips.","maker":"MME-Benchmarks team (Fu et al.)","year":2024,"metric":"Accuracy (tested with and without subtitles)","state":"active","sotaScore":"~89%","sotaModel":"Seed 2.1 Pro","sotaDate":"2026-06","leaderboardUrl":"https://github.com/MME-Benchmarks/Video-MME","sourceUrl":"https://arxiv.org/abs/2405.21075"},{"name":"LMArena","category":"preference-holistic","measures":"Crowdsourced human preference between two anonymized model responses, aggregated into a relative ranking, not an objective capability.","maker":"Arena (formerly LMArena and LMSYS Chatbot Arena; Angelopoulos, Chiang et al.)","year":2023,"metric":"Elo / Bradley-Terry pairwise rating (an Arena Score)","state":"active","sotaScore":"~1510 Elo","sotaModel":"Claude Opus 4.8","sotaDate":"2026-06","leaderboardUrl":"https://arena.ai/leaderboard","sourceUrl":"https://arxiv.org/abs/2403.04132","url":"https://capitalandcompute.net/ai-benchmarks/lmarena/"},{"name":"MT-Bench","category":"preference-holistic","measures":"Instruction-following and conversational quality on multi-turn prompts, scored automatically by a strong LLM judge.","maker":"LMSYS (Zheng et al., UC Berkeley)","year":2023,"metric":"LLM-as-judge score (1 to 10 scale, averaged)","state":"saturated","leaderboardUrl":"https://llm-stats.com/benchmarks/mt-bench","sourceUrl":"https://arxiv.org/abs/2306.05685"},{"name":"Artificial Analysis Intelligence Index","category":"preference-holistic","measures":"A composite index of overall model intelligence aggregating performance across reasoning, coding, knowledge, science and agentic tasks.","maker":"Artificial Analysis (independent)","year":2024,"metric":"Composite index score (0 to 100 aggregate); scores are not comparable across index versions","state":"active","sotaScore":"53 (index v4.3)","sotaModel":"Claude Fable 5.1 (max effort) and GPT-6 Astra (max)","sotaDate":"2026-09","leaderboardUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-intelligence-index","sourceUrl":"https://artificialanalysis.ai/methodology/intelligence-benchmarking","url":"https://capitalandcompute.net/ai-benchmarks/artificial-analysis-intelligence-index/"},{"name":"Vals Index","category":"preference-holistic","measures":"A composite of agentic finance, coding, legal and tax tasks, weighted by each sector's share of US GDP, meant to estimate the economic impact of a model rather than its general intelligence.","maker":"Vals AI (independent)","year":2025,"metric":"Weighted accuracy (%), with standard-error bars, plus cost per test and latency; scores are not comparable across index versions","state":"active","sotaScore":"68.90% (v2.1)","sotaModel":"Gemini 4 Argon (high effort)","sotaDate":"2026-09","leaderboardUrl":"https://www.vals.ai/benchmarks/vals_index","sourceUrl":"https://www.vals.ai/benchmarks/vals_index","url":"https://capitalandcompute.net/ai-benchmarks/vals-index/"},{"name":"HELM","category":"preference-holistic","measures":"Multi-metric holistic evaluation across many scenarios, reporting accuracy alongside calibration, robustness, fairness, bias, toxicity and efficiency.","maker":"Stanford CRFM (Liang, Bommasani et al.)","year":2022,"metric":"Multi-metric (per-metric scores across scenarios; no single headline number)","state":"active","leaderboardUrl":"https://crfm.stanford.edu/helm/","sourceUrl":"https://arxiv.org/abs/2211.09110"},{"name":"Artificial Analysis Coding Agent Index","category":"preference-holistic","measures":"Overall coding-agent capability as a single number, scoring the full stack (a specific model plus its harness and settings) rather than a model in isolation.","maker":"Artificial Analysis (independent)","year":2026,"metric":"Simple average of the component benchmark scores, with every task equally weighted","state":"active","leaderboardUrl":"https://artificialanalysis.ai/agents/coding-agents","sourceUrl":"https://artificialanalysis.ai/methodology/coding-agents-benchmarking"},{"name":"GDPval","category":"preference-holistic","measures":"Whether a model can produce the actual deliverables of skilled professional work (documents, slides, spreadsheets, diagrams) well enough to stand against an industry expert's version.","maker":"OpenAI, with an agentic re-run by Artificial Analysis as GDPval-AA","year":2025,"metric":"Blind pairwise comparison of two anonymised outputs on the same task, aggregated into an Elo rating; the v2 scale anchors human expert deliverables at 1000","state":"active","sotaScore":"1,764 Elo (GDPval-AA v2)","sotaModel":"Claude Fable 5.1 (adaptive reasoning, max effort)","sotaDate":"2026-09","leaderboardUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceUrl":"https://arxiv.org/abs/2510.04374","url":"https://capitalandcompute.net/ai-benchmarks/gdpval/"},{"name":"TruthfulQA","category":"safety-factuality","measures":"Whether a model avoids repeating common human misconceptions when answering questions, rather than imitating popular falsehoods.","maker":"Lin, Hilton, Evans (Oxford and OpenAI)","year":2021,"metric":"% truthful (and % truthful-and-informative)","state":"active","sourceUrl":"https://arxiv.org/abs/2109.07958","url":"https://capitalandcompute.net/ai-benchmarks/truthfulqa/"},{"name":"HaluEval","category":"safety-factuality","measures":"A model's ability to recognize hallucinated content across question answering, knowledge-grounded dialogue and summarization.","maker":"Li et al. (Renmin University of China)","year":2023,"metric":"Hallucination-recognition accuracy (faithful vs hallucinated)","state":"active","sourceUrl":"https://arxiv.org/abs/2305.11747"},{"name":"Vectara Hallucination Leaderboard","category":"safety-factuality","measures":"How often a model introduces unsupported content when summarizing a provided source document, i.e. faithfulness in closed-book summarization.","maker":"Vectara (Hughes et al.)","year":2023,"metric":"Hallucination rate (% of summaries judged unfaithful; lower is better)","state":"active","sotaScore":"1.8% (lower is better)","sotaModel":"antgroup/finix-s1-32b","sotaDate":"2026-05","leaderboardUrl":"https://github.com/vectara/hallucination-leaderboard","sourceUrl":"https://github.com/vectara/hallucination-leaderboard"},{"name":"SWE-bench-Live","category":"coding-agentic","measures":"The same real-GitHub-issue resolution task as SWE-bench, but on tasks harvested continuously from issues created after a model was trained.","maker":"Zhang, He, Zhang et al.","year":2025,"metric":"% resolved (pass@1)","state":"active","leaderboardUrl":"https://swe-bench-live.github.io/","sourceUrl":"https://arxiv.org/abs/2505.23419"},{"name":"SWE-rebench","category":"coding-agentic","measures":"Issue resolution on a continuously refreshed, decontaminated pool of Python software-engineering tasks mined automatically from open-source repositories.","maker":"Badertdinov, Golubev, Nekrashevich et al.","year":2025,"metric":"% resolved (pass@1)","state":"active","leaderboardUrl":"https://swe-rebench.com/","sourceUrl":"https://arxiv.org/abs/2505.20411"},{"name":"SWE-PolyBench","category":"coding-agentic","measures":"Whether a coding agent can resolve repository-level tasks outside Python, across Java, JavaScript, TypeScript and Python.","maker":"Rashid, Bock, Zhuang et al.","year":2025,"metric":"% resolved, plus syntax-tree-based retrieval and file-localisation metrics","state":"active","sourceUrl":"https://arxiv.org/abs/2504.08703"},{"name":"SciCode","category":"coding-agentic","measures":"Whether a model can write code that solves real scientific research problems, not general software tasks.","maker":"Tian, Gao, Zhang et al.","year":2024,"metric":"% of subproblems and main problems solved","state":"active","leaderboardUrl":"https://scicode-bench.github.io/","sourceUrl":"https://arxiv.org/abs/2407.13168"},{"name":"KernelBench","category":"coding-agentic","measures":"Whether a model can write GPU kernels that are both correct and actually faster than the PyTorch baseline.","maker":"Ouyang, Guo, Arora et al.","year":2025,"metric":"fast_p: % of generated kernels that are correct and at least p times faster than baseline","state":"active","sourceUrl":"https://arxiv.org/abs/2502.10517"},{"name":"PaperBench","category":"coding-agentic","measures":"Whether an agent can replicate a published AI research paper from scratch: understand the contribution, build the codebase, and run the experiments.","maker":"OpenAI (Starace, Jaffe, Sherburn et al.)","year":2025,"metric":"Replication score against a hierarchical rubric, graded by an LLM judge","state":"active","sourceUrl":"https://arxiv.org/abs/2504.01848"},{"name":"Commit0","category":"coding-agentic","measures":"Whether an agent can write an entire Python library from scratch against an API specification and an interactive test suite.","maker":"Zhao, Jiang, Lee et al.","year":2024,"metric":"% of unit tests passed","state":"active","leaderboardUrl":"https://commit-0.github.io/","sourceUrl":"https://arxiv.org/abs/2412.01769"},{"name":"EvalPlus","category":"coding-agentic","measures":"The same function-synthesis task as HumanEval and MBPP, rescored against far larger automatically generated test suites.","maker":"Liu, Xia, Wang et al.","year":2023,"metric":"pass@1 under the extended tests","state":"saturated","leaderboardUrl":"https://evalplus.github.io/leaderboard.html","sourceUrl":"https://arxiv.org/abs/2305.01210"},{"name":"BFCL","category":"agentic-tooluse","measures":"Whether a model calls functions and APIs correctly: picking the right function, filling parameters with valid types, and refusing to invent functions that were not offered.","maker":"UC Berkeley Gorilla team","year":2024,"metric":"Abstract-syntax-tree match against a reference call, plus executable checks; overall score is the unweighted mean of subcategories","state":"active","leaderboardUrl":"https://gorilla.cs.berkeley.edu/leaderboard.html","sourceUrl":"https://gorilla.cs.berkeley.edu/leaderboard.html","url":"https://capitalandcompute.net/ai-benchmarks/bfcl/"},{"name":"Mind2Web 2","category":"agentic-tooluse","measures":"Whether an agentic search or deep-research system can browse the live web and return a correct, citation-backed answer to a long-horizon question.","maker":"Gou, Huang, Ning et al.","year":2025,"metric":"Agent-as-a-Judge rubric scoring of answer correctness and citation support","state":"active","leaderboardUrl":"https://osu-nlp-group.github.io/Mind2Web-2/","sourceUrl":"https://arxiv.org/abs/2506.21506"},{"name":"WebVoyager","category":"agentic-tooluse","measures":"Whether a multimodal web agent can complete a user instruction end to end on real, live websites rather than a simulator or a static snapshot.","maker":"He, Yao, Ma et al.","year":2024,"metric":"Task success rate, judged automatically from screenshots and responses","state":"saturated","sourceUrl":"https://arxiv.org/abs/2401.13919"},{"name":"TheAgentCompany","category":"agentic-tooluse","measures":"Whether an agent can do real knowledge work inside a simulated software company: browsing, coding, using internal tools, and messaging simulated colleagues.","maker":"Xu, Song, Li et al.","year":2024,"metric":"Full and partial task completion, scored by checkpoint","state":"active","leaderboardUrl":"https://the-agent-company.com/","sourceUrl":"https://arxiv.org/abs/2412.14161"},{"name":"AndroidWorld","category":"agentic-tooluse","measures":"Whether an agent can operate a real Android phone to finish tasks across everyday apps.","maker":"Rawles, Clinckemaillie, Chang et al.","year":2024,"metric":"Programmatic reward from the device end state","state":"active","leaderboardUrl":"https://google-research.github.io/android_world/","sourceUrl":"https://arxiv.org/abs/2405.14573"},{"name":"Windows Agent Arena","category":"agentic-tooluse","measures":"Whether a multimodal agent can operate a full Windows desktop across the applications people actually use at work.","maker":"Bonatti, Zhao, Bonacci et al.","year":2024,"metric":"Task success rate from the OS end state","state":"active","sourceUrl":"https://arxiv.org/abs/2409.08264"},{"name":"Vending-Bench","category":"agentic-tooluse","measures":"Whether an agent stays coherent over a very long horizon, by running a simulated vending-machine business: stock, orders, pricing and daily fees.","maker":"Andon Labs (Backlund and Petersson)","year":2025,"metric":"Net worth and units sold at the end of the run","state":"active","sourceUrl":"https://arxiv.org/abs/2502.15840"},{"name":"LoCoMo","category":"agentic-tooluse","measures":"Whether an agent remembers and reasons over a conversation that spans months, rather than a single session.","maker":"Maharana, Lee, Tulyakov et al.","year":2024,"metric":"Question-answering accuracy, event summarisation and multi-modal dialogue generation","state":"active","sourceUrl":"https://arxiv.org/abs/2402.17753"},{"name":"ZebraLogic","category":"reasoning","measures":"Logical deduction under hard constraints, using logic grid puzzles generated from constraint-satisfaction problems.","maker":"Lin, Le Bras, Richardson et al.","year":2025,"metric":"Puzzle-level accuracy (all cells correct)","state":"active","sourceUrl":"https://arxiv.org/abs/2502.01100"},{"name":"EnigmaEval","category":"reasoning","measures":"Long multimodal puzzle solving: finding hidden connections between unrelated pieces of information and chaining many deductive steps.","maker":"Scale AI (Wang, Lee, Menghini et al.)","year":2025,"metric":"Exact-match accuracy on the final puzzle answer","state":"active","leaderboardUrl":"https://scale.com/leaderboard/enigma_eval","sourceUrl":"https://arxiv.org/abs/2502.08859"},{"name":"AGIEval","category":"reasoning","measures":"Human-centric reasoning, using questions from real standardised exams taken by people rather than synthetic datasets.","maker":"Zhong, Cui, Guo et al.","year":2023,"metric":"Accuracy","state":"saturated","sourceUrl":"https://arxiv.org/abs/2304.06364"},{"name":"DROP","category":"reasoning","measures":"Reading comprehension that requires discrete operations over a passage: resolving references then adding, counting or sorting.","maker":"Dua, Wang, Dasigi et al.","year":2019,"metric":"F1 and exact match","state":"saturated","sourceUrl":"https://arxiv.org/abs/1903.00161"},{"name":"PutnamBench","category":"math","measures":"Whether a neural theorem prover can produce a formal, machine-checked proof of an undergraduate competition problem.","maker":"Tsoukalas, Lee, Jennings et al.","year":2024,"metric":"% of theorems formally proved and machine-verified","state":"active","leaderboardUrl":"https://trishullab.github.io/PutnamBench/leaderboard.html","sourceUrl":"https://arxiv.org/abs/2407.11214"},{"name":"miniF2F","category":"math","measures":"Formal theorem proving on Olympiad-level mathematics, as a shared benchmark across proof assistants.","maker":"Zheng, Han and Polu","year":2021,"metric":"% of statements formally proved","state":"saturated","sourceUrl":"https://arxiv.org/abs/2109.00110"},{"name":"SuperGPQA","category":"knowledge","measures":"Graduate-level knowledge and reasoning across 285 disciplines, including the applied and service fields that mainstream benchmarks ignore.","maker":"M-A-P Team (Du, Yao et al.)","year":2025,"metric":"Accuracy","state":"active","sourceUrl":"https://arxiv.org/abs/2502.14739"},{"name":"HellaSwag","category":"knowledge","measures":"Commonsense sentence completion: picking the plausible continuation of an everyday scenario.","maker":"Zellers, Holtzman, Bisk et al.","year":2019,"metric":"Accuracy","state":"retired","sourceUrl":"https://arxiv.org/abs/1905.07830","url":"https://capitalandcompute.net/ai-benchmarks/hellaswag/"},{"name":"TriviaQA","category":"knowledge","measures":"Factual recall and reading comprehension over trivia questions with evidence documents.","maker":"Joshi, Choi, Weld et al.","year":2017,"metric":"Exact match and F1","state":"retired","sourceUrl":"https://arxiv.org/abs/1705.03551"},{"name":"IFEval","category":"instruction-multilingual","measures":"Whether a model obeys instructions that can be checked by a program, such as a word count, a required keyword, or a forbidden format.","maker":"Zhou, Lu, Mishra et al.","year":2023,"metric":"Strict and loose instruction-following accuracy","state":"saturated","sourceUrl":"https://arxiv.org/abs/2311.07911"},{"name":"Multi-IF","category":"instruction-multilingual","measures":"Whether a model keeps following instructions across multiple turns and in languages other than English.","maker":"He, Jin, Wang et al.","year":2024,"metric":"Instruction-following accuracy per turn","state":"active","sourceUrl":"https://arxiv.org/abs/2410.15553"},{"name":"Global-MMLU","category":"instruction-multilingual","measures":"Multilingual academic knowledge, separating questions that are culturally neutral from those requiring culture-specific knowledge.","maker":"Singh, Romanou, Fourrier et al.","year":2024,"metric":"Accuracy","state":"active","sourceUrl":"https://arxiv.org/abs/2412.03304"},{"name":"MGSM","category":"instruction-multilingual","measures":"Grade-school math word problems solved via chain-of-thought reasoning in ten languages.","maker":"Shi, Suzgun, Freitag et al.","year":2022,"metric":"Accuracy","state":"saturated","sourceUrl":"https://arxiv.org/abs/2210.03057"},{"name":"INCLUDE","category":"instruction-multilingual","measures":"Multilingual understanding built from local exam material, so the questions test regional knowledge rather than translated Western content.","maker":"Romanou, Foroutan, Sotnikova et al.","year":2024,"metric":"Accuracy","state":"active","sourceUrl":"https://arxiv.org/abs/2411.19799"},{"name":"LOFT","category":"long-context","measures":"Whether a long-context model can replace a retrieval pipeline outright: doing retrieval, RAG and SQL-style tasks natively from context.","maker":"Lee, Chen, Dai et al.","year":2024,"metric":"Task-specific accuracy compared against specialised retrieval pipelines","state":"active","sourceUrl":"https://arxiv.org/abs/2406.13121"},{"name":"HELMET","category":"long-context","measures":"Long-context ability across a wide spread of realistic downstream applications rather than one synthetic retrieval task.","maker":"Yen, Gao, Hou et al.","year":2024,"metric":"Per-category task metrics, reported across context lengths","state":"active","leaderboardUrl":"https://princeton-nlp.github.io/HELMET/","sourceUrl":"https://arxiv.org/abs/2410.02694"},{"name":"BABILong","category":"long-context","measures":"Reasoning over facts deliberately scattered through an extremely long document, not just retrieving one of them.","maker":"Kuratov, Bulatov, Anokhin et al.","year":2024,"metric":"Accuracy by context length","state":"active","sourceUrl":"https://arxiv.org/abs/2406.10149"},{"name":"MMBench","category":"multimodal","measures":"Fine-grained vision-language ability across a structured taxonomy of perception and reasoning skills.","maker":"Liu, Duan, Zhang et al.","year":2023,"metric":"Accuracy under circular evaluation","state":"saturated","leaderboardUrl":"https://mmbench.opencompass.org.cn/leaderboard","sourceUrl":"https://arxiv.org/abs/2307.06281"},{"name":"ChartQA","category":"multimodal","measures":"Question answering over charts that requires both reading visual features and doing arithmetic on them.","maker":"Masry, Long, Tan et al.","year":2022,"metric":"Relaxed accuracy (numeric answers within a tolerance)","state":"saturated","sourceUrl":"https://arxiv.org/abs/2203.10244"},{"name":"DocVQA","category":"multimodal","measures":"Question answering over scanned document images, where layout and structure carry the meaning.","maker":"Mathew, Karatzas and Jawahar","year":2020,"metric":"ANLS (average normalised Levenshtein similarity)","state":"saturated","sourceUrl":"https://arxiv.org/abs/2007.00398"},{"name":"CharXiv","category":"multimodal","measures":"Chart understanding on real, messy scientific figures rather than clean template-generated charts.","maker":"Wang, Xia, He et al.","year":2024,"metric":"Accuracy, split into descriptive and reasoning questions","state":"active","leaderboardUrl":"https://charxiv.github.io/","sourceUrl":"https://arxiv.org/abs/2406.18521"},{"name":"MMStar","category":"multimodal","measures":"Genuinely vision-dependent multimodal ability, on samples selected so the answer cannot be inferred from the text alone.","maker":"Chen, Li, Dong et al.","year":2024,"metric":"Accuracy, reported alongside a multimodal-gain and multimodal-leakage measure","state":"active","sourceUrl":"https://arxiv.org/abs/2403.20330"},{"name":"HealthBench","category":"domain-professional","measures":"Open-ended clinical conversation quality and safety, graded against rubrics written by practising physicians.","maker":"OpenAI (Arora, Wei, Soskin Hicks et al.)","year":2025,"metric":"Rubric score, graded by a model grader against physician-written criteria","state":"active","sourceUrl":"https://arxiv.org/abs/2505.08775","url":"https://capitalandcompute.net/ai-benchmarks/healthbench/"},{"name":"MedQA","category":"domain-professional","measures":"Medical knowledge, using real questions from professional medical board examinations.","maker":"Jin, Pan, Oufattole et al.","year":2020,"metric":"Accuracy","state":"saturated","sourceUrl":"https://arxiv.org/abs/2009.13081"},{"name":"MedHELM","category":"domain-professional","measures":"Clinical ability across the breadth of real medical work, on a clinician-validated taxonomy rather than exam questions.","maker":"Stanford CRFM","year":2025,"metric":"Per-task clinical metrics plus head-to-head win rates","state":"active","leaderboardUrl":"https://crfm.stanford.edu/helm/medhelm/latest/","sourceUrl":"https://arxiv.org/abs/2505.23802"},{"name":"LegalBench","category":"domain-professional","measures":"Legal reasoning across the specific skills lawyers actually use, as defined by legal professionals.","maker":"Guha, Nyarko, Ho et al.","year":2023,"metric":"Per-task accuracy, aggregated by reasoning type","state":"active","leaderboardUrl":"https://hazyresearch.stanford.edu/legalbench/","sourceUrl":"https://arxiv.org/abs/2308.11462","url":"https://capitalandcompute.net/ai-benchmarks/legalbench/"},{"name":"FinanceBench","category":"domain-professional","measures":"Open-book financial question answering over real public-company filings, with the supporting evidence required.","maker":"Patronus AI (Islam, Kannappan, Kiela et al.)","year":2023,"metric":"Answer correctness against the evidence, human reviewed","state":"active","sourceUrl":"https://arxiv.org/abs/2311.11944","url":"https://capitalandcompute.net/ai-benchmarks/financebench/"},{"name":"LiveBench","category":"preference-holistic","measures":"Broad capability across six categories at once (math, coding, reasoning, data analysis, instruction following, language), on questions refreshed monthly.","maker":"White, Dooley, Roberts et al.","year":2024,"metric":"Objective automatic scoring against ground truth, averaged across categories","state":"active","leaderboardUrl":"https://livebench.ai/","sourceUrl":"https://arxiv.org/abs/2406.19314","url":"https://capitalandcompute.net/ai-benchmarks/livebench/"},{"name":"Epoch Capabilities Index","category":"preference-holistic","measures":"Overall model capability on one continuous scale, stitched together from many benchmarks of differing difficulty.","maker":"Epoch AI","year":2025,"metric":"ECI score on the anchored scale","state":"active","leaderboardUrl":"https://epoch.ai/eci","sourceUrl":"https://epoch.ai/data/eci-documentation","url":"https://capitalandcompute.net/ai-benchmarks/epoch-capabilities-index/"},{"name":"METR Time Horizon","category":"preference-holistic","measures":"Model capability expressed in human time: the length of task, measured by how long humans take, that a model completes with 50% success.","maker":"METR (Kwa, West, Becker et al.)","year":2025,"metric":"50%-task-completion time horizon, in minutes or hours","state":"active","sourceUrl":"https://arxiv.org/abs/2503.14499","url":"https://capitalandcompute.net/ai-benchmarks/metr-time-horizon/"},{"name":"Arena-Hard-Auto","category":"preference-holistic","measures":"Human-preference-aligned quality on hard open-ended prompts, scored automatically instead of by live human voting.","maker":"Li, Chiang, Frick et al.","year":2024,"metric":"Win rate against a baseline model, judged by an LLM","state":"active","sourceUrl":"https://arxiv.org/abs/2406.11939"},{"name":"AlpacaEval 2 (Length-Controlled)","category":"preference-holistic","measures":"Instruction-following quality judged by an LLM, with a regression correction for the judge’s bias toward longer answers.","maker":"Dubois, Galambosi, Liang et al.","year":2024,"metric":"Length-controlled win rate","state":"saturated","leaderboardUrl":"https://tatsu-lab.github.io/alpaca_eval/","sourceUrl":"https://arxiv.org/abs/2404.04475"},{"name":"Copilot Arena","category":"preference-holistic","measures":"Which coding model developers actually prefer, collected from paired completions inside a real editor rather than a chat window.","maker":"Chi, Chen, Angelopoulos et al.","year":2025,"metric":"Elo-style ranking from in-editor pairwise preferences","state":"active","sourceUrl":"https://arxiv.org/abs/2502.09328"},{"name":"Cybench","category":"security-adversarial","measures":"Whether an agent can autonomously solve professional capture-the-flag security tasks: finding a vulnerability and executing an exploit.","maker":"Zhang, Perry, Dulepet et al.","year":2024,"metric":"% of tasks and subtasks solved unassisted","state":"active","leaderboardUrl":"https://cybench.github.io/","sourceUrl":"https://arxiv.org/abs/2408.08926"},{"name":"CyberSecEval 3","category":"security-adversarial","measures":"Cybersecurity risk across eight areas, split between risk to third parties and risk to the developers and users of an application.","maker":"Meta (Wan, Nikolaidis, Song et al.)","year":2024,"metric":"Per-risk pass and failure rates, measured with and without guardrails","state":"active","sourceUrl":"https://arxiv.org/abs/2408.01605"},{"name":"WMDP","category":"security-adversarial","measures":"Proxy knowledge of hazardous biosecurity, cybersecurity and chemical-security material.","maker":"Li, Pan, Gopal et al.","year":2024,"metric":"Accuracy, where lower is the desired direction after unlearning","state":"active","leaderboardUrl":"https://www.wmdp.ai/","sourceUrl":"https://arxiv.org/abs/2403.03218"},{"name":"AgentHarm","category":"security-adversarial","measures":"Whether a tool-using agent refuses explicitly malicious multi-step tasks, and whether it stays capable enough to complete them once jailbroken.","maker":"Andriushchenko, Souly, Dziemian et al.","year":2024,"metric":"Refusal rate and post-jailbreak task-completion rate","state":"active","sourceUrl":"https://arxiv.org/abs/2410.09024"}]}