{"columns":["benchmarkSlug","benchmarkName","organisation","category","version","summary","unit","scoreDirection","methodology","rankingUse","weight","lifecycle","normalizationVersion","reliabilityMultiplier","contaminationRisk","evidenceClass","constituentBenchmarkSlugs","supersedesBenchmarkSlugs","sourceIds","primarySourceKey","primarySourceId","primarySourceTitle","primarySourceUrl","publishedAt","checkedAt","verificationStatus","notes"],"generatedAt":"2026-09-01","ledgerSchemaVersion":"1.0.0","projection":"public","recordCount":435,"rows":[["swe-marathon","SWE Marathon","Abundant AI","coding","v1.1","Multi-hour whole-project software engineering tasks with hidden checks and exploit scanning.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"medium","direct",null,null,"swe-marathon-site","production::swe-marathon-site","swe-marathon-site","SWE Marathon","https://www.swe-marathon.org/",null,"2026-08-01","source-qualified",null],["benchlm-appbench","App-Bench","AfterQuery","coding","2025","A six-task full-stack web-app benchmark that measures how much required functionality an AI builder or coding assistant delivers from one prompt without human code edits.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"App-Bench","https://appbench.ai/",null,"2026-08-01","registry-only",null],["benchlm-financearena","FinanceArena — FinanceQA Assumption-Based","AfterQuery","knowledge","2025","An AfterQuery benchmark of open-ended financial analysis that requires models to read financial data, make assumptions, and return exact answers.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"FinanceQA","https://arxiv.org/abs/2501.18062",null,"2026-08-01","registry-only",null],["benchlm-idebench","IDE-Bench","AfterQuery","coding","2026","An 80-task software-engineering benchmark across eight repositories that tests whether autonomous IDE agents can explore, edit, run, and verify code changes end to end.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"IDE-Bench","https://arxiv.org/abs/2601.20886",null,"2026-08-01","registry-only",null],["benchlm-marketbench","Market-Bench","AfterQuery","agents","2025","A quantitative-trading implementation benchmark that asks models to build backtesters under market-book liquidity and execution-delay constraints, then compares their outputs with a verifier.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Market-Bench","https://arxiv.org/abs/2512.12264",null,"2026-08-01","registry-only",null],["forte","FORTE","AGI-Eval-Official","agents","2026-08","General agent benchmark published by LongCat-2.0 with an official GitHub repository and in-house LongCat scores.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"FORTE repository","https://github.com/AGI-Eval-Official/FORTE",null,"2026-08-15","source-qualified",null],["rwsearch","RWSearch","AGI-Eval-Official","research","2026-08","Search-agent benchmark published by LongCat-2.0 with an official GitHub repository and in-house LongCat scores.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"RW-Search repository","https://github.com/AGI-Eval-Official/RW-Search",null,"2026-08-15","source-qualified",null],["aider-polyglot","Aider Polyglot","Aider","coding","polyglot","225 Exercism coding exercises across C++, Go, Java, JavaScript, Python, and Rust that measure instruction following and code editing success.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"medium","direct",null,null,"refresh-aider-leaderboard","production::refresh-aider-leaderboard","refresh-aider-leaderboard","Aider polyglot leaderboard","https://aider.chat/docs/leaderboards/",null,"2026-07-21","source-qualified",null],["benchlm-olmocr","olmOCR-Bench","Allen Institute for AI","multimodal","2025","An end-to-end document understanding benchmark over long, layout-rich PDFs with tables, equations, headers, footnotes, and multi-column flows.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"olmOCR-Bench","https://github.com/allenai/olmocr/tree/main/olmocr/bench",null,"2026-08-01","registry-only",null],["benchlm-webarenaverified","WebArena-Verified Browser Agent Benchmark","Amine El Hattami, Megh Thakkar, Nicolas Chapados, Christopher Pal","agents","2025","WebArena-Verified is an audited release of the WebArena browser-agent benchmark. It rechecks task descriptions, reference answers, and evaluators, and replaces nondeterministic judging with deterministic checks where possible.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"WebArena-Verified: A Fully Audited Benchmark for Web Agents","https://openreview.net/forum?id=94tlGxmqkN",null,"2026-08-01","registry-only",null],["vending-bench","Vending-Bench","Andon Labs","agents","2","Long-horizon agent evaluation simulating a year of vending-machine business operations, measuring planning, tool use, and economic coherence.",null,"higher",null,"reference",0,"archived",null,null,"low","direct",null,null,"vending-bench-official","production::vending-bench-official","vending-bench-official","Vending-Bench","https://andonlabs.com/evals/vending-bench",null,"2026-07-21","registry-only",null],["benchlm-organicchemistryv2","Anthropic Organic Chemistry V2 evaluation","Anthropic","knowledge","2026","Chemistry tasks covering spectroscopy, synthesis planning, reaction prediction, and chemical structure images.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-proteindesign","Anthropic Protein Design evaluation","Anthropic","knowledge","2026","Generates novel protein sequences under family, topology, globularity, and structural-motif constraints.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["biomystery-bench","BioMysteryBench","Anthropic","research","2026-07","Biology evaluation suite covering hard research tasks and a human-solved subset, as published in Anthropic model launch tables.",null,"higher",null,"reference",0,"active","percent-direct-v1",null,"unknown","direct",null,null,"anthropic-claude-opus-5-launch","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Claude Opus 5 launch evaluation table","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","registry-only",null],["benchlm-biomysterybenchhumandifficult","BioMysteryBench Human Difficult","Anthropic","knowledge","2026","Computational biology challenges with objective answers that remained unsolved by independent human experts.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-biomysterybenchhumansolvable","BioMysteryBench Human Solvable","Anthropic","knowledge","2026","Computational biology challenges that independent human experts were able to solve.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-draco","Data Research and Analysis with Complex Operations","Anthropic","agents","2026","Agentic data-analysis tasks scored against per-task rubrics at a 980K-token budget.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-healthbenchlengthadjusted","HealthBench length-adjusted score","Anthropic","knowledge","2026","HealthBench score after applying a verbosity penalty to model responses.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-healthbenchprofessionalraw","HealthBench Professional raw score","Anthropic","knowledge","2026","Raw score on physician-authored clinical consult, documentation, and research conversations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-healthbench","HealthBench raw score","Anthropic","knowledge","2026","Raw score on realistic multi-turn healthcare conversations graded against expert-written rubrics.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-imo2026","International Mathematical Olympiad 2026","Anthropic","mathematics","2026","Proof-based olympiad performance on all six IMO 2026 problems.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-mcpatlasclaimcoverage","MCP-Atlas mean claim coverage","Anthropic","agents","2026","Average coverage of required claims in answers produced during real-world MCP tool-use workflows.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-protocolstroubleshooting","Molecular Biology Protocols Troubleshooting","Anthropic","knowledge","2026","Detects and fixes errors in molecular-biology protocols using document, code, and web-search tools.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-multiagentbrowsecompprerelease","Multi-Agent BrowseComp — 10-agent team prerelease configuration","Anthropic","agents","2026","BrowseComp accuracy from ten collaborating Opus 5 agents on a pre-release model and unreleased effort configuration.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-proteingymhard","ProteinGym Hard","Anthropic","knowledge","2026","Predicts mutation effects by ranking mutant protein sequences against wild type and comparing against laboratory measurements.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-toolathlonverifiedavgturns","Toolathlon Verified average assistant turns","Anthropic","agents","2026","Average assistant turns per Toolathlon Verified trajectory.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-toolathlonverifiedpass3all","Toolathlon Verified Pass cubed","Anthropic","agents","2026","Fraction of Toolathlon Verified tasks solved in all three independent trials.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-toolathlonverifiedpass3","Toolathlon Verified Pass@3","Anthropic","agents","2026","Fraction of Toolathlon Verified tasks solved in at least one of three trials.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-kindbench","KindBench Psychological Safety Benchmark","Aphelion Labs","knowledge","2026","A behavioral benchmark that tests psychological safety across sixteen adversarial multi-turn conversations covering emotional safety, identity, sycophancy, and value integrity.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"KindBench methodology","https://www.kindbench.com/methodology",null,"2026-08-01","registry-only",null],["benchlm-ppbench","Pencil Puzzle Bench","Approximate Labs","reasoning","2026","A multi-step verifiable reasoning benchmark that evaluates whether models can solve pencil puzzles with unique solutions.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Pencil Puzzle Bench","https://arxiv.org/abs/2603.02119",null,"2026-08-01","registry-only",null],["benchlm-arcagi1","ARC-AGI-1 Semi-Private Evaluation","ARC Prize Foundation","reasoning","2026","ARC Prize fluid-intelligence benchmark using novel visual grid transformations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"ARC Prize leaderboard","https://arcprize.org/",null,"2026-08-01","registry-only",null],["arc-agi-2","ARC-AGI-2","ARC Prize Foundation","reasoning","2","A second-generation abstract reasoning benchmark based on novel grid transformation tasks.",null,"higher",null,"ranking",0.025007,"active","percent-direct-v1",1,"low","direct",null,null,null,null,null,"ARC-AGI benchmark","https://arcprize.org/arc-agi",null,"2026-08-01","source-qualified",null],["arc-agi-3","ARC-AGI-3","ARC Prize Foundation","reasoning","3","An interactive reasoning benchmark testing exploration, world modelling, goal discovery, planning and adaptive execution in novel environments.",null,"higher",null,"not-weighted",0,"rolling",null,null,"low","direct",null,null,null,null,null,"ARC-AGI-3 official benchmark","https://arcprize.org/arc-agi/3",null,"2026-08-01","registry-only",null],["benchlm-aime2025arcee","AIME25 first-party comparison snapshot","Arcee AI","mathematics","2026","A display-only AIME25 reference from Arcee AI's Trinity-Large-Thinking launch chart.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","https://www.arcee.ai/blog/trinity-large-thinking",null,"2026-08-01","registry-only",null],["benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","Arcee AI","agents","2026","A function-calling benchmark for tool selection, schema adherence, and argument correctness.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","https://www.arcee.ai/blog/trinity-large-thinking",null,"2026-08-01","registry-only",null],["benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","Arcee AI","knowledge","2026","A display-only MMLU-Pro reference from Arcee AI's Trinity-Large-Thinking launch chart.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","https://www.arcee.ai/blog/trinity-large-thinking",null,"2026-08-01","registry-only",null],["benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","Arcee AI","coding","2026","A display-only SWE-bench Verified reference from Arcee AI's Trinity-Large-Thinking comparison chart.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","https://www.arcee.ai/blog/trinity-large-thinking",null,"2026-08-01","registry-only",null],["aa-automation-bench","AA AutomationBench","Artificial Analysis","agents",null,"Independently evaluated automation and agentic workflow benchmark from Artificial Analysis.",null,"higher",null,"ranking",0.016366,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"AA AutomationBench","https://benchlm.ai/benchmarks/aaAutomationBench",null,"2026-08-01","source-qualified",null],["aa-lcr","AA Long Context Reasoning","Artificial Analysis","reasoning",null,"Artificial Analysis long-context reasoning benchmark measuring extraction and synthesis over documents from roughly 10k to 100k tokens.",null,"higher",null,"ranking",0.015332,"active","percent-direct-v1",1,"low","direct",null,null,"aa-evaluation-artificial-analysis-long-context-reasoning-2026-08-27","production::aa-evaluation-artificial-analysis-long-context-reasoning-2026-08-27","aa-evaluation-artificial-analysis-long-context-reasoning-2026-08-27","AA-LCR Leaderboard","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning",null,"2026-08-01","source-qualified",null],["aa-briefcase","AA-Briefcase","Artificial Analysis","agents","2026","Artificial Analysis private agentic knowledge-work evaluation with Elo over rubric pass rate, analytical quality, and presentation quality on business deliverables.",null,"higher",null,"not-weighted",0,"active",null,null,"low","direct",null,null,"aa-briefcase","production::aa-briefcase","aa-briefcase","AA-Briefcase evaluation leaderboard","https://artificialanalysis.ai/evaluations/aa-briefcase",null,"2026-07-16","source-qualified",null],["aa-omniscience","AA-Omniscience","Artificial Analysis","research",null,"Artificial Analysis knowledge and hallucination benchmark measuring factual recall across economically relevant domains.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"unknown","composite","aa-omniscience-accuracy|aa-omniscience-non-hallucination",null,"aa-omniscience-leaderboard","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-08-01","source-qualified",null],["benchlm-aaagenticindex","Artificial Analysis Agentic Index","Artificial Analysis","agents","2026","A display-only Artificial Analysis agentic index.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","composite","gdpval-aa|tau3-banking|terminal-bench",null,"refresh-aa-model-leaderboard","production::refresh-aa-model-leaderboard","refresh-aa-model-leaderboard","Artificial Analysis model leaderboards","https://artificialanalysis.ai/leaderboards/models",null,"2026-08-01","registry-only",null],["benchlm-aaaime2025","Artificial Analysis AIME 2025","Artificial Analysis","mathematics","2026","An independently evaluated AIME 2025 result from Artificial Analysis.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Artificial Analysis AIME 2025 Benchmark Leaderboard","https://artificialanalysis.ai/evaluations/aime-2025",null,"2026-08-01","registry-only",null],["benchlm-aabriefcaseelo","Artificial Analysis Briefcase","Artificial Analysis","agents","2026","An independently evaluated professional-work benchmark reported as Elo.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"aa-briefcase","production::aa-briefcase","aa-briefcase","Artificial Analysis Briefcase Benchmark Leaderboard","https://artificialanalysis.ai/evaluations/aa-briefcase",null,"2026-08-01","registry-only",null],["benchlm-aacodingagents","Artificial Analysis Coding Agent Index","Artificial Analysis","coding","2026","A display-only Artificial Analysis leaderboard for coding-agent systems, combining agent harnesses, host models, and execution settings across software-engineering benchmarks.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Artificial Analysis Coding Agent Benchmarks","https://artificialanalysis.ai/agents/coding-agents",null,"2026-08-01","registry-only",null],["benchlm-aacodingindex","Artificial Analysis Coding Index","Artificial Analysis","coding","2026","A display-only Artificial Analysis coding index.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","composite","swe-bench-verified|scicode",null,"refresh-aa-model-leaderboard","production::refresh-aa-model-leaderboard","refresh-aa-model-leaderboard","Artificial Analysis model leaderboards","https://artificialanalysis.ai/leaderboards/models",null,"2026-08-01","registry-only",null],["benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","Artificial Analysis","agents","2026","An independently evaluated enterprise-operations benchmark from Artificial Analysis.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"aa-enterpriseops-gym","production::aa-enterpriseops-gym","aa-enterpriseops-gym","Artificial Analysis EnterpriseOps-Gym Benchmark Leaderboard","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-08-01","registry-only",null],["benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","Artificial Analysis","knowledge","2026","An independently evaluated multilingual knowledge result from Artificial Analysis.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"aa-global-mmlu-lite","production::aa-global-mmlu-lite","aa-global-mmlu-lite","Artificial Analysis Global-MMLU-Lite Benchmark Leaderboard","https://artificialanalysis.ai/evaluations/global-mmlu-lite","2026-07-15","2026-08-01","registry-only",null],["benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","Artificial Analysis","agents","2026","An independently evaluated legal-agent benchmark from Artificial Analysis.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"aa-harvey-lab","production::aa-harvey-lab","aa-harvey-lab","Artificial Analysis Harvey LAB-AA Benchmark Leaderboard","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-08-01","registry-only",null],["benchlm-aaitbench","Artificial Analysis ITBench-AA","Artificial Analysis","agents","2026","An independently evaluated IT-operations benchmark from Artificial Analysis.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"aa-itbench","production::aa-itbench","aa-itbench","Artificial Analysis ITBench-AA Benchmark Leaderboard","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-08-01","registry-only",null],["benchlm-aalivecodebench","Artificial Analysis LiveCodeBench","Artificial Analysis","coding","2026","An independently evaluated LiveCodeBench result from Artificial Analysis.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Artificial Analysis LiveCodeBench Benchmark Leaderboard","https://artificialanalysis.ai/evaluations/livecodebench",null,"2026-08-01","registry-only",null],["benchlm-aamath500","Artificial Analysis MATH-500","Artificial Analysis","mathematics","2026","An independently evaluated MATH-500 result from Artificial Analysis.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"aa-math-500","production::aa-math-500","aa-math-500","Artificial Analysis MATH-500 Benchmark Leaderboard","https://artificialanalysis.ai/evaluations/math-500","2026-07-15","2026-08-01","registry-only",null],["benchlm-aammlupro","Artificial Analysis MMLU-Pro","Artificial Analysis","knowledge","2026","An independently evaluated MMLU-Pro result from Artificial Analysis.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Artificial Analysis MMLU-Pro Benchmark Leaderboard","https://artificialanalysis.ai/evaluations/mmlu-pro",null,"2026-08-01","registry-only",null],["benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","Artificial Analysis","knowledge","2026","A display-only Artificial Analysis knowledge metric for the proportion of correctly answered questions.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"aa-grok-4-3-evals","production::aa-grok-4-3-evals","aa-grok-4-3-evals","Artificial Analysis model benchmarks","https://artificialanalysis.ai/models/grok-4-3",null,"2026-08-01","registry-only",null],["benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","Artificial Analysis","knowledge","2026","A display-only Artificial Analysis factuality metric for the rate of incorrect answers among non-correct responses.",null,"lower",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"aa-grok-4-3-evals","production::aa-grok-4-3-evals","aa-grok-4-3-evals","Artificial Analysis model benchmarks","https://artificialanalysis.ai/models/grok-4-3",null,"2026-08-01","registry-only",null],["benchlm-aaopennessindex","Artificial Analysis Openness Index","Artificial Analysis","knowledge","2026","A display-only Artificial Analysis model-openness index.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Artificial Analysis Openness Index","https://artificialanalysis.ai/evaluations/artificial-analysis-openness-index",null,"2026-08-01","registry-only",null],["benchlm-aatau3banking","Artificial Analysis Tau3-Banking","Artificial Analysis","agents","2026","An independently evaluated Tau3 banking benchmark from Artificial Analysis.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"aa-evaluation-tau3-banking-2026-08-27","production::aa-evaluation-tau3-banking-2026-08-27","aa-evaluation-tau3-banking-2026-08-27","Artificial Analysis Tau3-Banking Benchmark Leaderboard","https://artificialanalysis.ai/evaluations/tau3-banking",null,"2026-08-01","registry-only",null],["benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","Artificial Analysis","coding","2026","An independently evaluated Terminal-Bench v2.1 result from Artificial Analysis.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"aa-evaluation-terminalbench-v2-1-2026-08-27","production::aa-evaluation-terminalbench-v2-1-2026-08-27","aa-evaluation-terminalbench-v2-1-2026-08-27","Artificial Analysis Terminal-Bench v2.1 Benchmark Leaderboard","https://artificialanalysis.ai/evaluations/terminalbench-v2-1",null,"2026-08-01","registry-only",null],["benchlm-gdpvalaanormalized","GDPval-AA normalized","Artificial Analysis","agents","2026","A display-only Artificial Analysis normalized score for economically valuable tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"aa-grok-4-3-evals","production::aa-grok-4-3-evals","aa-grok-4-3-evals","Artificial Analysis model benchmarks","https://artificialanalysis.ai/models/grok-4-3",null,"2026-08-01","registry-only",null],["harvey-lab-aa","Harvey LAB-AA","Artificial Analysis / Harvey","agents",null,"Legal agent benchmark on private Harvey legal work tasks graded criterion-by-criterion in an agent sandbox.",null,"higher",null,"ranking",0.008518,"active","percent-direct-v1",1,"unknown","direct",null,null,"aa-harvey-lab","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","source-qualified",null],["itbench-aa","ITBench-AA","Artificial Analysis / IBM","agents",null,"Artificial Analysis implementation of IBM ITBench for Kubernetes incident root-cause analysis from offline snapshots.",null,"higher",null,"ranking",0.008518,"active","percent-direct-v1",1,"unknown","direct",null,null,"aa-itbench","production::aa-itbench","aa-itbench","ITBench-AA","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","source-qualified",null],["apex-agents","APEX-Agents-AA","Artificial Analysis / Mercor","agents",null,"Long-horizon professional multi-application agent benchmark implemented independently by Artificial Analysis.",null,"higher",null,"ranking",0.010221,"active","percent-direct-v1",1,"unknown","direct",null,null,"aa-apex-agents","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-08-01","source-qualified",null],["gdpval-aa","GDPval-AA v2","Artificial Analysis / OpenAI","agents","v2","Artificial Analysis agentic evaluation of OpenAI GDPval real-world economic work tasks across occupations, scored with pairwise Elo then normalized.",null,"higher",null,"ranking",0.017036,"active","percent-direct-v1",1,"low","direct",null,null,"aa-evaluation-gdpval-aa-2026-08-27","production::aa-evaluation-gdpval-aa-2026-08-27","aa-evaluation-gdpval-aa-2026-08-27","GDPval-AA v2 Leaderboard","https://artificialanalysis.ai/evaluations/gdpval-aa",null,"2026-08-01","source-qualified",null],["gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","Artificial Analysis / OpenAI","agents","v2","The source-native pairwise Elo track for Artificial Analysis' GDPval-AA v2 evaluation, kept separate from the normalized percentage track to prevent unit mixing.",null,"higher",null,"reference",0,"active","gdpval-aa-v2-elo-linear-v1",null,"low","direct",null,null,"aa-evaluation-gdpval-aa-2026-08-27","production::aa-evaluation-gdpval-aa-2026-08-27","aa-evaluation-gdpval-aa-2026-08-27","GDPval-AA v2","https://artificialanalysis.ai/evaluations/gdpval-aa",null,"2026-08-12","source-qualified",null],["enterpriseops-gym-aa","EnterpriseOps-Gym-AA","Artificial Analysis / ServiceNow","agents",null,"Stateful multi-step enterprise workflow agent benchmark graded on final database state across business domains.",null,"higher",null,"ranking",0.008518,"active","percent-direct-v1",1,"unknown","direct",null,null,"aa-enterpriseops-gym","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","source-qualified",null],["benchlm-voxpopuliwer","VoxPopuli-Cleaned-AA Word Error Rate","Artificial Analysis / VoxPopuli dataset authors","multimodal","2026","A speech-recognition benchmark on the cleaned Artificial Analysis VoxPopuli subset, reported as word error rate where lower is better.",null,"lower",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"VoxPopuli-Cleaned-AA","https://huggingface.co/datasets/ArtificialAnalysis/VoxPopuli-Cleaned-AA",null,"2026-08-01","registry-only",null],["automationbench-private","AutomationBench Private Set","AutomationBench","agents","Private set, August 2026","The private AutomationBench enterprise-workflow set, kept separate from the public 600-task track and reference-only by default.",null,"higher",null,"reference",0,"active","percent-direct-v1",null,"low","direct",null,null,"refresh-google-gemini-3-7-flash-evaluation","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Gemini 3.7 Flash model evaluation","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf",null,"2026-08-13","source-qualified",null],["balrog","BALROG","BALROG","agents",null,"Agent decision-making benchmark across long-horizon game and planning environments stressing exploration and credit assignment.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"low","direct",null,null,"registry-balrog","production::registry-balrog","registry-balrog","BALROG","https://balrogai.com/",null,"2026-07-21","registry-only",null],["benchlm-protocolsunderstanding","Benchling Molecular Biology Protocols Understanding","Benchling and Anthropic","knowledge","2026","Extends online molecular-biology protocols in additional directions using document and web-search tools.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["bl-agents","BenchLM Agentic prior","BenchLM","agents","bench-align-v5.1","Category prior from the BenchLM overall leaderboard agentic score used to estimate missing agentic coverage.",null,"higher",null,"reference",0,"active","percent-direct-v1",0.55,"unknown","composite",null,null,"benchlm-leaderboard","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM leaderboard","https://benchlm.ai/","2026-07-15","2026-07-15","source-qualified",null],["bl-coding","BenchLM Coding prior","BenchLM","coding","bench-align-v5.1","Category prior from the BenchLM overall leaderboard coding score used to estimate missing coding coverage.",null,"higher",null,"reference",0,"active","percent-direct-v1",0.55,"unknown","composite",null,null,"benchlm-leaderboard","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM leaderboard","https://benchlm.ai/","2026-07-15","2026-07-15","source-qualified",null],["bl-research","BenchLM Knowledge prior","BenchLM","research","bench-align-v5.1","Category prior from the BenchLM knowledge score used to estimate missing research coverage.",null,"higher",null,"reference",0,"active","percent-direct-v1",0.55,"unknown","composite",null,null,"benchlm-leaderboard","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM leaderboard","https://benchlm.ai/","2026-07-15","2026-07-15","source-qualified",null],["bl-mathematics","BenchLM Math prior","BenchLM","mathematics","bench-align-v5.1","Category prior from the BenchLM math score used to estimate missing mathematics coverage.",null,"higher",null,"reference",0,"active","percent-direct-v1",0.55,"unknown","composite",null,null,"benchlm-leaderboard","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM leaderboard","https://benchlm.ai/","2026-07-15","2026-07-15","source-qualified",null],["bl-multimodal","BenchLM Multimodal prior","BenchLM","multimodal","bench-align-v5.1","Category prior from the BenchLM multimodal score used to estimate missing multimodal coverage.",null,"higher",null,"reference",0,"active","percent-direct-v1",0.55,"unknown","composite",null,null,"benchlm-leaderboard","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM leaderboard","https://benchlm.ai/","2026-07-15","2026-07-15","source-qualified",null],["bl-reasoning","BenchLM Reasoning prior","BenchLM","reasoning","bench-align-v5.1","Category prior from the BenchLM overall leaderboard reasoning score used to estimate missing reasoning coverage.",null,"higher",null,"reference",0,"active","percent-direct-v1",0.55,"unknown","composite",null,null,"benchlm-leaderboard","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM leaderboard","https://benchlm.ai/","2026-07-15","2026-07-15","source-qualified",null],["benchlm-koreancsat","College Scholastic Ability Test (수능)","BenchLM registry","knowledge",null,"The Korean SAT exam.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"College Scholastic Ability Test (수능)","https://benchlm.ai/benchmarks/koreanCsat",null,"2026-08-01","registry-only",null],["benchlm-click","Cultural and Linguistic Intelligence in Korean","BenchLM registry","knowledge",null,"Evaluates Korean culture and linguistics.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Cultural and Linguistic Intelligence in Korean","https://benchlm.ai/benchmarks/click",null,"2026-08-01","registry-only",null],["benchlm-gaia","General AI Assistants","BenchLM registry","agents","2024","GAIA evaluates AI models on real-world tasks that are conceptually simple for humans but require multi-step reasoning, web browsing, tool use, and multimodal understanding for AI. Tasks span three difficulty levels and test practical assistant capabilities rather than academic knowledge.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,"benchlm-gaia","production::benchlm-gaia","benchlm-gaia","General AI Assistants","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-08-01","registry-only",null],["benchlm-hrm8k","HAE-RAE Math 8K","BenchLM registry","knowledge",null,"Korean mathematical reasoning (high-school to Olympiad level).",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"HAE-RAE Math 8K","https://benchlm.ai/benchmarks/hrm8k",null,"2026-08-01","registry-only",null],["benchlm-ifbench","Instruction Following Benchmark","BenchLM registry","instruction-following","2025","IFBench evaluates precise instruction-following generalization on 58 challenging, verifiable out-of-domain constraints. Unlike IFEval which tests familiar constraint types, IFBench specifically measures how well models follow novel instructions they haven't been optimized for, exposing overfitting to common instruction patterns.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Instruction Following Benchmark","https://benchlm.ai/benchmarks/ifBench",null,"2026-08-01","registry-only",null],["benchlm-kmmlupro","KMMLU-Pro","BenchLM registry","knowledge",null,"Korean National Professional Licensure exams evaluating professional-grade knowledge.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"KMMLU-Pro","https://benchlm.ai/benchmarks/kmmluPro",null,"2026-08-01","registry-only",null],["benchlm-kmmluredux","KMMLU-Redux","BenchLM registry","knowledge",null,"Cleaned KMMLU from national technical qualification exams, with errors removed, decontaminated, and deduplicated.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"KMMLU-Redux","https://benchlm.ai/benchmarks/kmmluRedux",null,"2026-08-01","registry-only",null],["benchlm-kobalt","Korean Benchmark for Advanced Linguistic Tasks","BenchLM registry","knowledge",null,"Evaluates advanced Korean linguistic competence.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Korean Benchmark for Advanced Linguistic Tasks","https://benchlm.ai/benchmarks/kobalt",null,"2026-08-01","registry-only",null],["benchlm-scicode","Scientific Code Benchmark","BenchLM registry","coding","2024","SciCode evaluates language models on generating code for realistic scientific research problems across 16 subfields of physics, math, chemistry, biology, and material science. Problems decompose into 338 subproblems requiring domain knowledge recall, scientific reasoning, and precise code synthesis. Based on real scripts from published research.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Scientific Code Benchmark","https://benchlm.ai/benchmarks/sciCode",null,"2026-08-01","registry-only",null],["bfcl-v4","BFCL v4","Berkeley","agents","v4","Berkeley Function-Calling Leaderboard v4 measuring structured tool and function invocation correctness across multi-turn scenarios.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"low","direct",null,null,"refresh-bfcl","production::refresh-bfcl","refresh-bfcl","Berkeley Function Calling Leaderboard","https://gorilla.cs.berkeley.edu/leaderboard.html",null,"2026-07-21","registry-only",null],["berkeley-function-calling-leaderboard","Berkeley Function-Calling Leaderboard","Berkeley Gorilla","agents","V3","An evaluation suite for function selection, argument construction and tool-use behaviour.",null,"higher",null,"not-weighted",0,"rolling",null,null,"unknown","direct",null,null,"refresh-bfcl","production::refresh-bfcl","refresh-bfcl","Berkeley Function-Calling Leaderboard","https://gorilla.cs.berkeley.edu/leaderboard.html",null,"2026-07-14","registry-only",null],["bigcodebench","BigCodeBench","BigCodeBench authors","coding",null,"A benchmark for practical code generation involving diverse libraries and complex instructions.",null,"higher",null,"not-weighted",0,"active",null,null,"unknown","direct",null,null,null,null,null,"BigCodeBench official repository","https://github.com/bigcode-project/bigcodebench",null,"2026-08-01","registry-only",null],["benchlm-claweval","Claw-Eval","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","agents","2026","A transparent real-world autonomous-agent benchmark with 300 human-verified tasks, 2,159 rubric items, and Pass^3 scoring across general, multi-turn, and native multimodal agent tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claw-Eval: Towards Trustworthy Evaluation of Autonomous Agents","https://arxiv.org/abs/2604.06132",null,"2026-08-01","registry-only",null],["browsecomp-plus","BrowseComp-Plus","BrowseComp","agents",null,"Harder web-browsing agent evaluation extending BrowseComp with longer multi-hop research and evidence synthesis tasks.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"low","direct",null,null,"registry-browsecomp-plus","production::registry-browsecomp-plus","registry-browsecomp-plus","BrowseComp","https://github.com/texttron/BrowseComp-Plus",null,"2026-07-21","registry-only",null],["benchlm-brumo2025","Bulgarian Mathematical Olympiad 2025","Bulgarian Mathematical Society","mathematics","2025","A challenging mathematical olympiad competition featuring problems that test advanced mathematical reasoning and problem-solving skills at the olympiad level.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Bulgarian Mathematical Olympiad","https://www.math.bas.bg/",null,"2026-08-01","registry-only",null],["benchlm-edgebench","EdgeBench","ByteDance Seed","coding","2026","A systems and software-engineering benchmark from ByteDance Seed that evaluates agents on long-horizon edge tasks using time-budgeted learning curves rather than a single static pass rate.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"EdgeBench","https://edge-bench.org/",null,"2026-08-01","registry-only",null],["benchlm-ceval","C-Eval","C-Eval authors","knowledge","2023","A Chinese-language academic and professional benchmark spanning humanities, social science, STEM, and applied subjects.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"C-Eval: A Multi-Level Multi-Discipline Chinese Evaluation Suite for Foundation Models","https://arxiv.org/abs/2305.08322",null,"2026-08-01","registry-only",null],["benchlm-reactnativeevals","React Native Evals","Callstack","coding","2026","An open benchmark for AI coding agents on real-world React Native implementation tasks, emphasizing working app behavior, recommended architecture choices, and strict constraint adherence.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"React Native Evals","https://rn-evals.vercel.app/",null,"2026-08-01","registry-only",null],["benchlm-caistextleaderboard","CAIS AI Dashboard Text Capabilities Index","Center for AI Safety","knowledge","2025","A Center for AI Safety dashboard view summarizing text capabilities across HLE, ARC-AGI-2, SWE-Bench Pro, and TextQuests.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"CAIS AI Dashboard","https://dashboard.safe.ai/",null,"2026-08-01","registry-only",null],["mask","MASK","Center for AI Safety","research",null,"Model honesty evaluation measuring whether models fabricate answers when uncertain versus appropriately abstaining.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"low","direct",null,null,"registry-mask","production::registry-mask","registry-mask","MASK benchmark","https://www.mask-benchmark.ai/",null,"2026-07-21","registry-only",null],["humanitys-last-exam","Humanity's Last Exam","Center for AI Safety and Scale AI","reasoning",null,"A broad expert-level academic benchmark covering difficult questions across many domains.",null,"higher",null,"ranking",0.025007,"rolling","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"Humanity's Last Exam","https://lastexam.ai/",null,"2026-08-01","source-qualified",null],["chartqa","ChartQA","ChartQA authors","multimodal",null,"A visual question-answering benchmark focused on charts and data visualisations.",null,"higher",null,"not-weighted",0,"archived",null,null,"unknown","direct",null,null,null,null,null,"ChartQA official repository","https://github.com/vis-nlp/ChartQA",null,"2026-07-14","registry-only",null],["charxiv","CharXiv","CharXiv","multimodal","May 2026","Measures reasoning over charts in scientific papers.",null,"higher",null,"ranking",0.041721,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"CharXiv","https://charxiv.github.io/",null,"2026-08-01","source-qualified",null],["benchlm-charxivnotools","CharXiv Reasoning without tools","CharXiv authors","multimodal","2024","Tool-free variant of CharXiv that isolates raw visual reasoning ability without code execution or tool augmentation.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"CharXiv: Charting Gaps in Realistic Chart Understanding in Multimodal LLMs","https://charxiv.github.io/",null,"2026-08-01","registry-only",null],["code-arena-webdev","Code Arena Web Development","Code Arena","coding","August 2026","The public Code Arena web-development Elo leaderboard retained as a reference-only model-comparison track.",null,"higher",null,"reference",0,"rolling","elo-1000-linear-v1",null,"unknown","direct",null,null,"refresh-google-gemini-3-7-flash-evaluation","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Gemini 3.7 Flash model evaluation","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf",null,"2026-08-13","source-qualified",null],["benchlm-frontiercode11extended","FrontierCode 1.1 Extended","Cognition","coding","2026","Cognition's 150-task Extended subset of the FrontierCode 1.1 software-engineering benchmark.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GPT-5.6 models are now available in Devin","https://devin.ai/blog/gpt-5-6",null,"2026-08-01","registry-only",null],["benchlm-frontiercode","FrontierCode 1.1 Main","Cognition","coding","2026","Cognition's 100-task software-engineering benchmark for whether coding agents produce mergeable, production-quality pull requests, scored for correctness, tests, scope, style, and maintainability through maintainer-authored rubrics.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"FrontierCode leaderboard","https://cognition.com/frontiercode",null,"2026-08-01","registry-only",null],["global-mmlu-lite","Global-MMLU-Lite","Cohere Labs","research",null,"Lightweight multilingual MMLU-style knowledge benchmark spanning diverse languages and cultural contexts.",null,"higher",null,"not-weighted",0,"archived","percent-direct-v1",1,"unknown","direct",null,null,"aa-global-mmlu-lite","production::aa-global-mmlu-lite","aa-global-mmlu-lite","Global-MMLU-Lite","https://artificialanalysis.ai/evaluations/global-mmlu-lite","2026-07-15","2026-07-16","source-qualified",null],["critpt","CritPt","CritPt authors","reasoning",null,"Research-level physics reasoning benchmark with composite challenges designed to stress frontier scientific reasoning.",null,"higher",null,"ranking",0.015332,"active","percent-direct-v1",1,"low","direct",null,null,"aa-evaluation-critpt-2026-08-27","production::aa-evaluation-critpt-2026-08-27","aa-evaluation-critpt-2026-08-27","CritPt Leaderboard","https://artificialanalysis.ai/evaluations/critpt",null,"2026-08-01","source-qualified",null],["cruxeval","CRUXEval","CRUXEval authors","coding",null,"A code-reasoning benchmark based on predicting program inputs and outputs.",null,"higher",null,"not-weighted",0,"archived",null,null,"unknown","direct",null,null,null,null,null,"CRUXEval official repository","https://github.com/facebookresearch/cruxeval",null,"2026-07-14","registry-only",null],["cursor-bench","CursorBench","Cursor","coding","3.2","Cursor's first-party benchmark for ambiguous multi-file coding-agent tasks from real Cursor sessions (v3.2 public snapshot).",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"medium","direct",null,null,"cursor-bench-public","production::cursor-bench-public","cursor-bench-public","CursorBench via BenchLM","https://benchlm.ai/benchmarks/cursorBench","2026-07-15","2026-08-01","source-qualified",null],["benchlm-kmmluhard","KMMLU-Hard","Daekeun ML","knowledge","2025","A filtered hard subset of KMMLU containing ~5,000 questions that most models get wrong.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Evaluating LLMs on Hard Korean Queries","https://github.com/daekeun-ml/evaluate-llm-on-korean-dataset",null,"2026-08-01","registry-only",null],["benchlm-math500","MATH-500 Problem Set","Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","mathematics","2021","A curated subset of 500 problems from the MATH dataset, covering algebra, counting and probability, geometry, intermediate algebra, number theory, prealgebra, and precalculus.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Measuring Mathematical Problem Solving With the MATH Dataset","https://arxiv.org/abs/2103.03874",null,"2026-08-01","registry-only",null],["benchlm-mmlu","Massive Multitask Language Understanding","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","knowledge","2020","A comprehensive multiple-choice question answering test covering 57 tasks including elementary mathematics, US history, computer science, law, and more. Tests knowledge across diverse academic subjects from high school to professional level.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Measuring Massive Multitask Language Understanding","https://arxiv.org/abs/2009.03300",null,"2026-08-01","registry-only",null],["benchlm-officeqa","OfficeQA","Databricks and Anthropic","multimodal","2026","Grounded numerical reasoning over a corpus of historical U.S. Treasury Bulletin documents.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["deepswe","DeepSWE","DataCurve","coding","v1.1","Contamination-resistant software engineering repair tasks across real repositories under a shared agent harness.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"low","direct",null,null,"deepswe-datacurve","production::deepswe-datacurve","deepswe-datacurve","DeepSWE leaderboard","https://deepswe.datacurve.ai/",null,"2026-08-01","source-qualified",null],["benchlm-gpqa","Graduate-Level Google-Proof Q&A","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","knowledge","2023","A challenging dataset of 448 multiple-choice questions written by domain experts in biology, physics, and chemistry. Designed to be difficult even for skilled non-experts with access to Google.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","https://arxiv.org/abs/2311.12022",null,"2026-08-01","registry-only",null],["benchlm-deepplanning","DeepPlanning","DeepPlanning authors","agents","2026","A long-horizon planning benchmark that tests whether agents can optimize under explicit time, budget, and feasibility constraints.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepPlanning: Benchmarking Long-Horizon Agentic Planning with Verifiable Constraints","https://arxiv.org/abs/2601.18137",null,"2026-08-01","registry-only",null],["benchlm-livecodebenchpass1cot","LiveCodeBench Pass@1 with Chain-of-Thought","DeepSeek","coding","2026","This lane contains DeepSeek's LiveCodeBench Pass@1-COT results. The explicit metric and prompting label keeps them separate from generic and version-specific LiveCodeBench rows.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 technical report","https://benchlm.ai/benchmarks/liveCodeBenchPass1Cot",null,"2026-08-01","registry-only",null],["benchlm-agieval","AGIEval","DeepSeek-AI","knowledge","2026","A human-centric exam benchmark for general knowledge and reasoning reported in DeepSeek-V4 base-model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/agieval",null,"2026-08-01","registry-only",null],["benchlm-apex","Apex","DeepSeek-AI","mathematics","2026","A high-difficulty mathematical reasoning benchmark reported in DeepSeek-V4 model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/apex",null,"2026-08-01","registry-only",null],["benchlm-apexshortlist","Apex Shortlist","DeepSeek-AI","mathematics","2026","A shortlist subset of the Apex mathematical reasoning benchmark reported in DeepSeek-V4 model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/apexShortlist",null,"2026-08-01","registry-only",null],["benchlm-cmmlu","Chinese Massive Multitask Language Understanding","DeepSeek-AI","knowledge","2026","A Chinese multitask academic benchmark reported in DeepSeek-V4 base-model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/cmmlu",null,"2026-08-01","registry-only",null],["benchlm-chinesesimpleqa","Chinese-SimpleQA","DeepSeek-AI","knowledge","2026","A Chinese short-form factuality benchmark reported by DeepSeek for V4 model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/chineseSimpleQa",null,"2026-08-01","registry-only",null],["benchlm-cluewsc","CLUEWSC","DeepSeek-AI","reasoning","2026","A Chinese Winograd Schema Challenge benchmark reported in DeepSeek-V4 base-model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/cluewsc",null,"2026-08-01","registry-only",null],["benchlm-cmath","CMath","DeepSeek-AI","mathematics","2026","A Chinese mathematical reasoning benchmark reported in DeepSeek-V4 base-model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/cmath",null,"2026-08-01","registry-only",null],["benchlm-codeforces","Codeforces Rating","DeepSeek-AI","coding","2026","Competitive-programming rating reported for DeepSeek-V4 thinking-mode evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/codeforces",null,"2026-08-01","registry-only",null],["benchlm-corpusqa1m","CorpusQA 1M","DeepSeek-AI","reasoning","2026","A million-token CorpusQA long-context question-answering benchmark reported in DeepSeek-V4 model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/corpusQa1m",null,"2026-08-01","registry-only",null],["benchlm-dsbenchfullstack","DeepSeek DSBench FullStack","DeepSeek-AI","coding","2026","DeepSeek's internal full-stack coding-agent benchmark.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"deepseek-v4-flash-vision-exp-release","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek V4 Flash 0731 update","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-01","registry-only",null],["benchlm-dsbenchhard","DeepSeek DSBench Hard","DeepSeek-AI","coding","2026","DeepSeek's internal hard coding-agent benchmark.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"deepseek-v4-flash-vision-exp-release","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek V4 Flash 0731 update","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-01","registry-only",null],["benchlm-drop","Discrete Reasoning Over Paragraphs","DeepSeek-AI","reasoning","2026","A reading-comprehension benchmark requiring discrete reasoning over paragraphs, reported in DeepSeek-V4 base-model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/drop",null,"2026-08-01","registry-only",null],["benchlm-factsparametric","FACTS Parametric","DeepSeek-AI","knowledge","2026","A parametric factuality benchmark reported in DeepSeek-V4 base-model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/factsParametric",null,"2026-08-01","registry-only",null],["benchlm-gsm8k","Grade School Math 8K","DeepSeek-AI","mathematics","2026","A grade-school mathematical reasoning benchmark reported in DeepSeek-V4 base-model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/gsm8k",null,"2026-08-01","registry-only",null],["benchlm-hellaswag","HellaSwag","DeepSeek-AI","reasoning","2026","A commonsense natural-language inference benchmark reported in DeepSeek-V4 base-model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/hellaswag",null,"2026-08-01","registry-only",null],["benchlm-hlewithtools","Humanity's Last Exam with tools","DeepSeek-AI","agents","2026","Tool-augmented Humanity's Last Exam scores reported in DeepSeek-V4 thinking-mode evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/hleWithTools",null,"2026-08-01","registry-only",null],["benchlm-imoanswerbench","IMOAnswerBench","DeepSeek-AI","mathematics","2026","A challenging mathematical reasoning benchmark reported in DeepSeek-V4 model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/imoAnswerBench",null,"2026-08-01","registry-only",null],["benchlm-mathbenchmark","MATH","DeepSeek-AI","mathematics","2026","A competition-style mathematical reasoning benchmark reported in DeepSeek-V4 base-model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/mathBenchmark",null,"2026-08-01","registry-only",null],["benchlm-mrcr1m","MRCR 1M","DeepSeek-AI","reasoning","2026","A million-token MRCR long-context retrieval benchmark reported in DeepSeek-V4 model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/mrcr1m",null,"2026-08-01","registry-only",null],["benchlm-multiloko","MultiLoKo","DeepSeek-AI","knowledge","2026","A multilingual/localized knowledge benchmark reported in DeepSeek-V4 base-model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/multiLoKo",null,"2026-08-01","registry-only",null],["terminal-bench-21-provider","Terminal-Bench 2.1 (provider run)","DeepSeek-AI","coding","2026","A provider-run Terminal-Bench 2.1 result stored separately from the repository's Terminal-Bench 2.0 lane.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"deepseek-v4-flash-vision-exp-release","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek V4 Flash 0731 update","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-01","registry-only",null],["benchlm-triviaqa","TriviaQA","DeepSeek-AI","knowledge","2026","A reading and trivia question-answering benchmark reported in DeepSeek-V4 base-model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/triviaQa",null,"2026-08-01","registry-only",null],["benchlm-winogrande","WinoGrande","DeepSeek-AI","reasoning","2026","A commonsense coreference benchmark reported in DeepSeek-V4 base-model evaluations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"DeepSeek-V4 Technical Report","https://benchlm.ai/benchmarks/winogrande",null,"2026-08-01","registry-only",null],["benchlm-designarenawebsite","Design Arena Website Elo","Design Arena","multimodal","2026","A display-only Design Arena website-generation Elo score surfaced on OpenRouter model benchmark pages.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"OpenRouter Grok 4.3 benchmarks","https://openrouter.ai/x-ai/grok-4.3/benchmarks",null,"2026-08-01","registry-only",null],["benchlm-designarenaagenticwebdev","Design Arena Agentic Web Dev Elo","Design Arena / Intelligence","agents","2026","A display-only Elo rating from blinded comparisons of multi-file web applications built by coding agents.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Design Arena Web Dev (Agentic) Leaderboard","https://intelligence.ai/leaderboard/webapps",null,"2026-08-01","registry-only",null],["docvqa","DocVQA","DocVQA organisers","multimodal",null,"A benchmark family for visual question answering over document images.",null,"higher",null,"not-weighted",0,"archived",null,null,"unknown","direct",null,null,null,null,null,"DocVQA challenge","https://rrc.cvc.uab.es/?ch=17",null,"2026-07-14","registry-only",null],["benchlm-kernelbench","KernelBench Hard H100","Elliot Arledge","coding","2026","An agentic GPU-kernel benchmark that measures how much of the hardware roofline a model's correct, audit-clean kernels reach on six demanding CUDA and Triton problems.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"KernelBench Hard","https://kernelbench.com/hard?gpu=h100",null,"2026-08-01","registry-only",null],["frontiermath","FrontierMath","Epoch AI","mathematics",null,"A benchmark of difficult, expert-authored mathematics problems designed for frontier systems.",null,"higher",null,"ranking",0.022969,"rolling","percent-direct-v1",1,"low","direct",null,null,"refresh-frontiermath-v1","production::refresh-frontiermath-v1","refresh-frontiermath-v1","FrontierMath","https://epoch.ai/frontiermath",null,"2026-08-01","source-qualified",null],["benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","Epoch AI","mathematics","2026","Epoch AI's corrected v2 Tier 4 expansion, a separate set of exceptionally difficult research-level mathematics problems evaluated with Python-enabled iterative reasoning.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"FrontierMath Tier 4 v2 leaderboard","https://epoch.ai/benchmarks/frontiermath-tier-4-v2?view=graph&tab=leaderboard",null,"2026-08-01","registry-only",null],["benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","Epoch AI","mathematics","2026","Epoch AI's corrected v2 core FrontierMath suite of private advanced mathematics problems. Models can reason iteratively and use Python; scores are pass rates on the private set.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"refresh-frontiermath-v2-tiers-1-3","production::refresh-frontiermath-v2-tiers-1-3","refresh-frontiermath-v2-tiers-1-3","FrontierMath Tiers 1-3 (v2)","https://epoch.ai/benchmarks/frontiermath-tiers-1-3-v2",null,"2026-08-16","registry-only",null],["eq-bench","EQ-Bench","EQ-Bench","instruction-following","3","Emotional intelligence and nuanced social reasoning benchmark for chat models using scored multi-turn vignettes.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"unknown","direct",null,null,"registry-eq-bench","production::registry-eq-bench","registry-eq-bench","EQ-Bench","https://eqbench.com/",null,"2026-07-21","registry-only",null],["benchlm-eqbench4","EQ-Bench 4","EQ-Bench","agents","2026","A multi-turn benchmark of emotional and social intelligence using synthetic personas and pairwise LLM judging.",null,"higher",null,"reference",0,"review-pending","field-relative-reference-v1",null,"unknown","direct",null,null,"registry-eq-bench","production::registry-eq-bench","registry-eq-bench","EQ-Bench 4","https://eqbench.com/",null,"2026-08-01","registry-only",null],["exploitbench","ExploitBench","ExploitBench authors","agents","2","A cybersecurity-agent benchmark measuring progress from reaching vulnerable code through exploit primitives and code execution.",null,"higher",null,"ranking",0.016366,"rolling","percent-direct-v1",1,"low","direct",null,null,null,null,null,"ExploitBench official repository","https://github.com/exploitbench/exploitbench",null,"2026-07-15","source-qualified",null],["exploitgym","ExploitGym","ExploitGym authors","agents","2026-05","A realistic cybersecurity benchmark for evaluating whether agents can turn known software vulnerabilities into concrete attacks.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"low","direct",null,null,null,null,null,"ExploitGym paper","https://arxiv.org/abs/2605.11086",null,"2026-08-01","source-qualified",null],["benchlm-mgsm","Multilingual Grade School Math","Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","knowledge","2022","A multilingual benchmark that translates 250 grade school math problems from GSM8K into 10 typologically diverse languages: Bengali, German, Spanish, French, Japanese, Russian, Swahili, Telugu, Thai, and Chinese.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Language Models are Multilingual Chain-of-Thought Reasoners","https://arxiv.org/abs/2210.03057",null,"2026-08-01","registry-only",null],["frontier-bench","Frontier-Bench","Frontier-Bench","agents","v0.1","Agentic terminal coding tasks scored as mean reward over repeated attempts under a shared mini-SWE-agent harness.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"Frontier-Bench","https://www.frontierbench.ai/",null,"2026-07-24","source-qualified",null],["frontierswe","FrontierSWE","FrontierSWE","coding","2026-07","Extremely difficult implementation, performance, and research software tasks scored with pairwise dominance.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"unknown","direct",null,null,"frontierswe-site","production::frontierswe-site","frontierswe-site","FrontierSWE","https://www.frontierswe.com/",null,"2026-08-01","source-qualified",null],["gaia","GAIA","GAIA authors","agents",null,"A benchmark for general assistants that must reason, browse, use tools and combine modalities.",null,"higher",null,"ranking",0.016366,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"GAIA benchmark collection","https://huggingface.co/gaia-benchmark",null,"2026-07-15","source-qualified",null],["benchlm-gertlabs","Gert Labs Composite Game Benchmark","Gert Labs","agents","2026","A game-environment benchmark that evaluates AI models in novel games covering strategic planning, resource management, spatial reasoning, cooperation, and theory of mind.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Gert Labs rankings","https://gertlabs.com/rankings",null,"2026-08-01","registry-only",null],["blueprint-bench-2","Blueprint-Bench 2","Google","multimodal","v2","Measures agentic spatial reasoning over blueprint-style visual information.",null,"higher",null,"reference",0,"active",null,null,"unknown","direct",null,null,"google-gemini-35-model-card","production::google-gemini-35-model-card","google-gemini-35-model-card","Blueprint-Bench 2","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-08-01","source-qualified",null],["finance-agent-v2","Finance Agent v2","Google","research","v2","Measures multi-step financial research and evidence synthesis by an agent system.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"unknown","direct",null,null,"google-gemini-35-model-card","production::google-gemini-35-model-card","google-gemini-35-model-card","Finance Agent v2","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-08-01","source-qualified",null],["mcp-atlas","MCP Atlas","Google","agents","May 2026","Tests an agent system's use of tools exposed through Model Context Protocol servers.",null,"higher",null,"ranking",0.016366,"active","percent-direct-v1",1,"unknown","direct",null,null,"google-gemini-35-model-card","production::google-gemini-35-model-card","google-gemini-35-model-card","MCP Atlas","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-08-01","source-qualified",null],["toolathlon","Toolathlon","Google","agents","May 2026","Measures multi-tool planning and execution across tool-use workloads.",null,"higher",null,"ranking",0.016366,"active","percent-direct-v1",1,"unknown","direct",null,null,"google-gemini-35-model-card","production::google-gemini-35-model-card","google-gemini-35-model-card","Toolathlon","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-08-01","source-qualified",null],["gdm-mrcr-v2-128k-average","GDM-MRCR v2 8-needle, 128K average","Google DeepMind","long-context","v2, 128K average","Google's cumulative eight-needle MRCR v2 score at a 128K average context length, retained as an exact reference track.",null,"higher",null,"reference",0,"active","percent-direct-v1",null,"unknown","direct",null,"mrcr-v2","refresh-google-gemini-3-7-flash-evaluation","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Gemini 3.7 Flash model evaluation","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf",null,"2026-08-13","source-qualified",null],["ifeval","IFEval","Google Research","instruction-following",null,"An evaluation of whether language models follow verifiable natural-language instructions.",null,"higher",null,"not-weighted",0,"active",null,null,"unknown","direct",null,null,null,null,null,"IFEval implementation","https://github.com/google-research/google-research/tree/master/instruction_following_eval",null,"2026-07-14","registry-only",null],["gpqa-diamond","GPQA Diamond","GPQA authors","reasoning",null,"The Diamond subset of a graduate-level, expert-written question-answering benchmark.",null,"higher",null,"ranking",0.025007,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"GPQA official repository","https://github.com/idavidrein/gpqa",null,"2026-08-01","source-qualified",null],["terminal-bench-3","Terminal-Bench 3.0","Harbor / Terminal-Bench","agents","3.0","Version 3.0 of the terminal-agent benchmark, retained as a reference-only family until methodology 1.8 defines a compatible scoring anchor.",null,"higher",null,"reference",0,"active","percent-direct-v1",null,"unknown","direct",null,"terminal-bench","xai-grok-4-6-release-2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6 release evaluation table","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","source-qualified",null],["terminal-bench-4","Terminal-Bench 4.0","Harbor / Terminal-Bench","agents","4.0.0","A breaking 66-task release of the continuously versioned terminal-agent benchmark with calibrated resources, a flat eight-hour agent timeout, eight task removals, and 19 task fixes.",null,"higher",null,"reference",0,"review-pending","percent-direct-v1",null,"unknown","direct",null,"terminal-bench-3","terminal-bench-4-release-2026-08-28","production::terminal-bench-4-release-2026-08-28","terminal-bench-4-release-2026-08-28","Terminal-Bench 4.0","https://www.tbench.ai/news/terminal-bench-4-0","2026-08-28","2026-08-30","source-qualified",null],["benchlm-hmmt2023","Harvard-MIT Mathematics Tournament February 2023","Harvard and MIT Mathematics Departments","mathematics","2023","A prestigious high school mathematics competition hosted jointly by Harvard and MIT, featuring challenging problems across various mathematical disciplines.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Harvard-MIT Mathematics Tournament","https://www.hmmt.org/",null,"2026-08-01","registry-only",null],["benchlm-hmmt2024","Harvard-MIT Mathematics Tournament February 2024","Harvard and MIT Mathematics Departments","mathematics","2024","The 2024 February edition of the Harvard-MIT Mathematics Tournament, continuing the tradition of challenging high school mathematics competition.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Harvard-MIT Mathematics Tournament","https://www.hmmt.org/",null,"2026-08-01","registry-only",null],["benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","Harvard and MIT Mathematics Departments","mathematics","2025","The most recent February edition of the Harvard-MIT Mathematics Tournament, featuring the latest challenging problems in competitive mathematics.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Harvard-MIT Mathematics Tournament","https://www.hmmt.org/",null,"2026-08-01","registry-only",null],["benchlm-legalagentbenchheldoutallpass","Legal Agent Benchmark all-pass rate — Harvey held-out set","Harvey AI","agents","2026","Harvey AI's strict held-out task success rate requiring every rubric criterion to pass.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-legalagentbenchheldoutcriterionpass","Legal Agent Benchmark mean criterion-pass rate — Harvey held-out set","Harvey AI","agents","2026","Harvey AI's mean criterion-level score on its held-out legal-agent evaluation.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-legalagentbenchallpass","Legal Agent Benchmark all-pass rate — Anthropic harness","Harvey AI and Anthropic","agents","2026","Strict task success requiring every expert-written legal-work rubric criterion to pass.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-legalagentbenchcriterionpass","Legal Agent Benchmark mean criterion-pass rate — Anthropic harness","Harvey AI and Anthropic","agents","2026","Mean fraction of expert-written rubric criteria passed across legal-agent tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-weirdml","WeirdML v2","Havard Tveit Ihle","knowledge","2026","A machine-learning engineering benchmark that tests whether LLMs can train models on novel datasets, write PyTorch code, and improve through iterative feedback.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"WeirdML","https://htihle.github.io/weirdml.html",null,"2026-08-01","registry-only",null],["hle-verified","HLE-Verified","HLE-Verified authors","reasoning","1,811-item verified set","The 1,811-item verified Humanity's Last Exam set comprising 668 verified original items and 1,143 revised items.",null,"higher",null,"reference",0,"active","percent-direct-v1",null,"low","direct",null,"humanitys-last-exam",null,null,null,"HLE-Verified","https://arxiv.org/abs/2602.13964",null,"2026-08-13","source-qualified",null],["hmmt-feb-2026","HMMT Feb 2026","HMMT","mathematics",null,"Harvard-MIT Mathematics Tournament February 2026 problems used as a contest math evaluation.",null,"higher",null,"ranking",0.02297,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"HMMT Feb 2026","https://benchlm.ai/benchmarks/hmmtFeb2026",null,"2026-07-15","source-qualified",null],["ifbench","IFBench","IFBench","instruction-following",null,"Instruction-following evaluation measuring constraint and format reliability.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"IFBench","https://benchlm.ai/benchmarks/ifBench",null,"2026-08-01","source-qualified",null],["benchlm-sobvalueacc","Structured Output Benchmark Value Accuracy","Interfaze","instruction-following","2026","A structured-output benchmark from Interfaze measuring whether extracted JSON leaf values exactly match verified ground truth.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Structured Output Benchmark Leaderboard","https://interfaze.ai/leaderboards/structured-output-benchmark",null,"2026-08-01","registry-only",null],["benchlm-researchclawbench","ResearchClawBench","InternScience","agents","2026","An end-to-end autonomous scientific research benchmark with 40 tasks across 10 scientific domains, where agents receive related literature and raw data, then attempt to rediscover the hidden target paper.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research","https://arxiv.org/abs/2606.07591",null,"2026-08-01","registry-only",null],["factorio-learning-env","Factorio Learning Environment","Jack Hopkins et al.","agents",null,"Long-horizon agentic engineering evaluation inside Factorio factories, testing planning, tool use, and multi-hour progress.",null,"higher",null,"reference",0,"archived",null,null,"low","direct",null,null,"registry-factorio-learning-env","production::registry-factorio-learning-env","registry-factorio-learning-env","Factorio Learning Environment","https://jackhopkins.github.io/factorio-learning-environment/",null,"2026-07-21","registry-only",null],["benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","knowledge","2024","A benchmark that evaluates the ability of language models to answer short, fact-seeking questions accurately. Focuses on factual correctness rather than reasoning complexity.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Measuring short-form factuality in large language models","https://arxiv.org/abs/2411.04368",null,"2026-08-01","registry-only",null],["benchlm-ifeval","Instruction-Following Eval","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","instruction-following","2023","A benchmark of 541 prompts built from 25 verifiable instruction types. It tests whether a model follows checkable constraints such as keyword, length, casing, and response-format requirements.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Instruction-Following Evaluation for Large Language Models","https://arxiv.org/abs/2311.07911",null,"2026-08-01","registry-only",null],["benchlm-inferencebench","InferenceBench","Jehyeok Yeon, Ben Rank, Maksym Andriushchenko","agents","2026","A benchmark for open-ended LLM inference optimization by AI agents. Agents receive a base model, one H100, and a fixed time budget to build a valid OpenAI-compatible inference server that improves serving speed.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"InferenceBench","https://inferencebench.ai/",null,"2026-08-01","registry-only",null],["benchlm-programbench","ProgramBench: Can Language Models Rebuild Programs From Scratch?","John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","coding","2026","A cleanroom software-engineering benchmark where agents receive only a compiled executable and documentation, then must architect and implement a complete codebase that reproduces the original program's behavior.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"ProgramBench: Can Language Models Rebuild Programs From Scratch?","https://programbench.com/static/paper.pdf",null,"2026-08-01","registry-only",null],["benchlm-pinchbench","PinchBench","Kilo Code","agents","2026","An OpenClaw agent benchmark from Kilo that measures successful task completion across standardized real-world agent workflows.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"About PinchBench","https://pinchbench.com/about",null,"2026-08-01","registry-only",null],["benchlm-kmmlu","Korean Massive Multitask Language Understanding","KMMLU Authors","knowledge","2024","Evaluates Korean expert-level knowledge across 45 subjects. 20% of questions require Korean cultural context.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"KMMLU: Measuring Massive Multitask Language Understanding in Korean","https://arxiv.org/abs/2402.11548",null,"2026-08-01","registry-only",null],["labbench2","LABBench2","LABBench","research","2","A biology real-world research-task evaluation run with a Linux terminal, scientific tools and restricted internet access.",null,"higher",null,"reference",0,"active","percent-direct-v1",null,"unknown","direct",null,null,"refresh-google-gemini-3-7-flash-evaluation","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Gemini 3.7 Flash model evaluation","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf",null,"2026-08-13","source-qualified",null],["benchlm-singlecellbench","LatchBio SingleCellBench","LatchBio and Anthropic","knowledge","2026","Single-cell RNA sequencing analysis tasks spanning common bioinformatics workflows.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-spatialbenchverified","LatchBio SpatialBench Verified","LatchBio and Anthropic","knowledge","2026","Analysis of spatial transcriptomics data across externally validated biological problems.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["terminal-bench-hard","Terminal-Bench Hard","Laude Institute / Artificial Analysis","agents",null,"Harder agentic terminal evaluation suite spanning software engineering, systems administration, and data processing tasks.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"low","direct",null,null,"aa-terminalbench-hard","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-08-01","source-qualified",null],["benchlm-liquidextractjsonvalidity","Liquid image-to-JSON extraction JSON validity","Liquid AI","multimodal","2026","A display-only Liquid AI extraction metric measuring the share of image-to-JSON outputs that parse as strict JSON.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"LiquidAI LFM2.5-VL Extract model cards","https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B-Extract",null,"2026-08-01","registry-only",null],["benchlm-liquidextractschemaf1","Liquid image-to-JSON extraction schema consistency F1","Liquid AI","multimodal","2026","A display-only Liquid AI extraction metric measuring field-name agreement between requested schema fields and extracted JSON fields.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"LiquidAI LFM2.5-VL Extract model cards","https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B-Extract",null,"2026-08-01","registry-only",null],["benchlm-liquidextractvlmjudge","Liquid image-to-JSON extraction VLM judge score","Liquid AI","multimodal","2026","A display-only Liquid AI extraction metric measuring judged agreement between extracted values and the source image.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"LiquidAI LFM2.5-VL Extract model cards","https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B-Extract",null,"2026-08-01","registry-only",null],["benchlm-mkqa11","MKQA-11 multilingual retrieval","Liquid AI","knowledge","2026","A display-only multilingual QA retrieval benchmark reported by Liquid AI for LFM2.5 retriever models, using Recall@20 across 11 languages.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"LFM2.5 Retrievers: Bi-directional LFMs for Fast Multilingual Search","https://www.liquid.ai/blog/lfm2-5-retrievers",null,"2026-08-01","registry-only",null],["benchlm-nanobeirmultilingual","NanoBEIR Multilingual Extended","Liquid AI","knowledge","2026","A display-only multilingual retrieval benchmark reported by Liquid AI for LFM2.5 retriever models, using NDCG@10 across 11 languages.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"LFM2.5 Retrievers: Bi-directional LFMs for Fast Multilingual Search","https://www.liquid.ai/blog/lfm2-5-retrievers",null,"2026-08-01","registry-only",null],["livebench","LiveBench","LiveBench team","reasoning",null,"A frequently refreshed benchmark suite designed to limit contamination and cover multiple capabilities.",null,"higher",null,"ranking",0.025007,"rolling","percent-direct-v1",1,"low","direct",null,null,"refresh-livebench","production::refresh-livebench","refresh-livebench","LiveBench","https://livebench.ai/","2026-06-25","2026-08-01","source-qualified",null],["benchlm-livecodebenchv5","LiveCodeBench v5","LiveCodeBench maintainers","coding","2025","LiveCodeBench v5 is a named release and date-window slice. BenchLM keeps explicitly labeled v5 rows outside the rolling weighted lane.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,"refresh-livecodebench-repo","production::refresh-livecodebench-repo","refresh-livecodebench-repo","LiveCodeBench official repository and release documentation","https://github.com/LiveCodeBench/LiveCodeBench",null,"2026-08-01","registry-only",null],["benchlm-livecodebenchv6","LiveCodeBench v6","LiveCodeBench maintainers","coding","2026","LiveCodeBench v6 is a named release slice used in provider comparison tables. Keeping it separate prevents v6 results from being mixed into older or rolling LiveCodeBench windows.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"refresh-livecodebench-repo","production::refresh-livecodebench-repo","refresh-livecodebench-repo","LiveCodeBench official repository and release documentation","https://github.com/LiveCodeBench/LiveCodeBench",null,"2026-08-01","registry-only",null],["benchlm-livecodebenchpro","LiveCodeBench Pro","LiveCodeBench Pro authors","coding","2025","A harder competitive-programming benchmark family built from Codeforces, ICPC, and IOI problems, with quarter-specific public leaderboards and difficulty-aware reporting.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"LiveCodeBench Pro: How Do Olympiad Medalists Judge LLMs in Competitive Programming?","https://arxiv.org/abs/2506.11928",null,"2026-08-01","registry-only",null],["livecodebench","LiveCodeBench","LiveCodeBench team","coding",null,"A continuously updated benchmark for code generation, execution, self-repair and related tasks.",null,"higher",null,"ranking",0.031694,"rolling","percent-direct-v1",1,"medium","direct",null,null,"refresh-livecodebench-repo","production::refresh-livecodebench-repo","refresh-livecodebench-repo","LiveCodeBench official repository","https://github.com/LiveCodeBench/LiveCodeBench",null,"2026-08-01","source-qualified",null],["arena-hard-v2","Arena-Hard v2","LMSYS","instruction-following","v2","Hard preference-style chat evaluation derived from Arena battle prompts, measuring instruction-following and helpfulness under automated judging.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"medium","direct",null,null,"registry-arena-hard-v2","production::registry-arena-hard-v2","registry-arena-hard-v2","Arena-Hard-Auto","https://github.com/lmarena/arena-hard-auto",null,"2026-07-21","registry-only",null],["arena-hard-auto","Arena-Hard-Auto","LMSYS Org","instruction-following",null,"An automated pairwise evaluation set derived from challenging user prompts.",null,"higher",null,"not-weighted",0,"rolling",null,null,"unknown","direct",null,null,null,null,null,"Arena-Hard-Auto official repository","https://github.com/lm-sys/arena-hard-auto",null,"2026-07-14","registry-only",null],["long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","Long-Horizon-Terminal-Bench authors","agents","2026-07","A 46-task terminal benchmark for long-running agent work with dense subtask grading across nine practical domains.",null,"higher",null,"ranking",0.016366,"active","percent-direct-v1",1,"low","direct",null,null,null,null,null,"Long-Horizon Terminal-Bench project","https://zli12321.github.io/LHTB/",null,"2026-07-15","source-qualified",null],["lvbench","LVBench","LVBench","multimodal","August 2026","A long-video understanding benchmark for retrieving and reasoning over information distributed across extended video inputs.",null,"higher",null,"reference",0,"active","percent-direct-v1",null,"unknown","direct",null,null,"refresh-google-gemini-3-7-flash-evaluation","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Gemini 3.7 Flash model evaluation","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf",null,"2026-08-13","source-qualified",null],["aime-2025","AIME 2025","MAA / public contest evals","mathematics",null,"American Invitational Mathematics Examination 2025 contest problems used as a frontier math evaluation.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"AIME 2025","https://benchlm.ai/benchmarks/aime2025",null,"2026-08-01","source-qualified",null],["aime-2026","AIME 2026","MAA / public contest evals","mathematics",null,"American Invitational Mathematics Examination 2026 contest problems used as a frontier math evaluation.",null,"higher",null,"ranking",0.022969,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"AIME 2026","https://benchlm.ai/benchmarks/aime2026",null,"2026-08-01","source-qualified",null],["usamo-2026","USAMO 2026","MAA / public contest evals","mathematics",null,"United States of America Mathematical Olympiad 2026 problems used as an expert math evaluation.",null,"higher",null,"ranking",0.022969,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"USAMO 2026","https://benchlm.ai/benchmarks/usamo2026",null,"2026-08-01","source-qualified",null],["benchlm-humaneval","Evaluating Large Language Models Trained on Code","Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","coding","2021","A set of 164 handwritten Python function-generation problems. HumanEval is useful as a historical floor check, but BenchLM's current exact-source table is too small to support a broad frontier-coding verdict.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Evaluating Large Language Models Trained on Code","https://arxiv.org/abs/2107.03374",null,"2026-08-01","registry-only",null],["benchlm-autocadbench","AutoCAD-Bench","Markov Studios","agents","2026","A Markov Studios computer-use benchmark that asks agents to produce 2D drawings and 3D models in AutoCAD.",null,"higher",null,"reference",0,"review-pending","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"AutoCAD-Bench","https://www.markovstudios.com/research/autocad-bench",null,"2026-08-01","registry-only",null],["matharena-apex","MathArena Apex","MathArena","mathematics","Apex","Hard mathematical reasoning suite used in frontier model cards to stress multi-step symbolic problem solving beyond saturated contest sets.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"MathArena","https://matharena.ai/",null,"2026-07-21","registry-only",null],["benchlm-arxivmathjune2026withtools","ArXivMath June 2026 with tools","MathArena and Anthropic","mathematics","2026","Final-answer research mathematics problems drawn from recent arXiv abstracts with tool access.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-arxivmathjune2026","ArXivMath June 2026 without tools","MathArena and Anthropic","mathematics","2026","Final-answer research mathematics problems drawn from recent arXiv abstracts.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-aime2023","American Invitational Mathematics Examination 2023","Mathematical Association of America","mathematics","2023","A 15-question, 3-hour examination where each answer is an integer from 000 to 999. Serves as the intermediate step between AMC 10/12 and the USA Mathematical Olympiad (USAMO).",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"American Invitational Mathematics Examination","https://www.maa.org/math-competitions/aime",null,"2026-08-01","registry-only",null],["benchlm-aime2024","American Invitational Mathematics Examination 2024","Mathematical Association of America","mathematics","2024","The 2024 edition of AIME, maintaining the same format of 15 challenging mathematics problems with integer answers from 000 to 999.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"American Invitational Mathematics Examination","https://www.maa.org/math-competitions/aime",null,"2026-08-01","registry-only",null],["mathvista","MathVista","MathVista authors","multimodal",null,"A benchmark for mathematical reasoning over diagrams, charts and other visual inputs.",null,"higher",null,"not-weighted",0,"archived",null,null,"unknown","direct",null,null,"refresh-mathvista","production::refresh-mathvista","refresh-mathvista","MathVista","https://mathvista.github.io/",null,"2026-07-14","registry-only",null],["benchlm-runescapebench","RuneBench / runescape-bench","Max Bittker","knowledge","2026","An agentic coding benchmark where models use a TypeScript SDK to play a RuneScape-like environment and optimize skill-training performance.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"RuneBench","https://maxbittker.github.io/runebench/",null,"2026-08-01","registry-only",null],["benchlm-mcpmarkverified","MCPMark-Verified","MCPMark","agents","2026","A human-verified edition of MCPMark for MCP tool use across Notion, GitHub, Filesystem, Postgres, and Playwright server environments.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MCPMark","https://mcpmark.ai/",null,"2026-08-01","registry-only",null],["benchlm-vitabench","VITA-Bench","Meituan LongCat Team","agents","2025","An interactive real-world agent benchmark grounded in practical consumer-service tasks such as delivery, in-store consumption, and online travel workflows.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"VitaBench: Benchmarking LLM Agents with Versatile Interactive Tasks in Real-world Applications","https://vitabench.github.io/",null,"2026-08-01","registry-only",null],["benchlm-osworld2","OSWorld 2.0","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","agents","2026","A long-horizon computer-use benchmark covering realistic workflows across everyday and professional desktop tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"OSWorld2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks","https://arxiv.org/abs/2606.29537",null,"2026-08-01","registry-only",null],["apex-swe","APEX-SWE","Mercor","coding",null,"Mercor's professional software-engineering benchmark, added as a reference-only family pending owner-version reconciliation and explicit methodology treatment.",null,"higher",null,"reference",0,"active","percent-direct-v1",null,"unknown","direct",null,null,null,null,null,"APEX-SWE leaderboard","https://www.mercor.com/apex/apex-swe-leaderboard/",null,"2026-08-12","source-qualified",null],["meta-internal-coding-bench","Meta Internal Coding Bench","Meta","coding","August 2026","A Meta-internal coding benchmark of 440 tasks derived from the Meta codebase and evaluated in an internal harness.",null,"higher",null,"not-weighted",0,"review-pending","percent-direct-v1",1,"unknown","direct",null,null,"meta-muse-spark-1-2-methodology-2026-08-05","production::meta-muse-spark-1-2-methodology-2026-08-05","meta-muse-spark-1-2-methodology-2026-08-05","Muse Spark 1.2 Evaluation Methodology","https://research.meta.ai/static/muse-spark-1-2-methodology","2026-08-05","2026-08-05","registry-only",null],["benchlm-babyvision","BabyVision","Meta AI","multimodal","2026","A multimodal benchmark for fine-grained visual perception and grounded reasoning tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"meta-muse-spark-1-1-eval","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Muse Spark 1.1 Evaluation Report","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-08-01","registry-only",null],["benchlm-deepsearchqa","DeepSearchQA","Meta AI","agents","2026","An agentic browsing benchmark where models search the web, gather evidence, and answer list-style questions using browser tools.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Muse Spark Eval Methodology","https://ai.meta.com/static-resource/muse-spark-eval-methodology",null,"2026-08-01","registry-only",null],["benchlm-frontierscienceresearch","FrontierScience Research","Meta AI","knowledge","2026","A research-focused FrontierScience evaluation variant for scientific investigation and problem solving.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Muse Spark Eval Methodology","https://ai.meta.com/static-resource/muse-spark-eval-methodology",null,"2026-08-01","registry-only",null],["benchlm-ipho2025theory","International Physics Olympiad 2025 (Theory)","Meta AI","mathematics","2026","The three official theory problems from the 2025 International Physics Olympiad, scored with blinded human evaluation.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Muse Spark Eval Methodology","https://ai.meta.com/static-resource/muse-spark-eval-methodology",null,"2026-08-01","registry-only",null],["benchlm-medxpertqamm","MedXpertQA Multimodal","Meta AI","multimodal","2026","A multimodal medical multiple-choice benchmark covering clinical images such as X-rays, histology, and dermatology.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Muse Spark Eval Methodology","https://ai.meta.com/static-resource/muse-spark-eval-methodology",null,"2026-08-01","registry-only",null],["benchlm-medxpertqatext","MedXpertQA Text","Meta AI","knowledge","2026","A medical multiple-choice benchmark spanning many specialties with 10 answer options per question.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Muse Spark Eval Methodology","https://ai.meta.com/static-resource/muse-spark-eval-methodology",null,"2026-08-01","registry-only",null],["benchlm-reactbench","ReactBench v1","Million","coding","2026","A coding-agent benchmark for realistic React work, with rubrics that check production concerns such as performance, accessibility, correctness, and code quality.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"ReactBench","https://www.reactbench.com/",null,"2026-08-01","registry-only",null],["benchlm-bankertoolbench","BankerToolBench","MiniMax","agents","2026","A display-only provider benchmark for finance-oriented tool-use and agent workflows.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MiniMax M3 model card","https://huggingface.co/MiniMaxAI/MiniMax-M3",null,"2026-08-01","registry-only",null],["benchlm-gdpvalrubrics","GDPval rubrics","MiniMax","agents","2026","A display-only provider-table GDPval rubric score for economically valuable work tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MiniMax M3 model card","https://huggingface.co/MiniMaxAI/MiniMax-M3",null,"2026-08-01","registry-only",null],["benchlm-kernelbenchhard","KernelBench Hard","MiniMax","coding","2026","A display-only benchmark for difficult GPU kernel implementation and optimization tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MiniMax M3 model card","https://huggingface.co/MiniMaxAI/MiniMax-M3",null,"2026-08-01","registry-only",null],["benchlm-mlebenchlite","MLE-Bench Lite","MiniMax","agents","2026","A lightweight machine-learning competition benchmark that measures whether models can iteratively train, evaluate, and improve ML systems in low-resource settings.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MiniMax M2.7: Early Echoes of Self-Evolution","https://www.minimax.io/news/minimax-m27-en",null,"2026-08-01","registry-only",null],["benchlm-mmclawbench","MM-ClawBench","MiniMax","agents","2026","An OpenClaw-derived agent benchmark covering practical work and life tasks such as office document delivery, research, planning, and code maintenance.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MiniMax M2.7: Early Echoes of Self-Evolution","https://www.minimax.io/news/minimax-m27-en",null,"2026-08-01","registry-only",null],["benchlm-nl2repo","NL2Repo","MiniMax","coding","2026","A repository-understanding benchmark that measures whether models can map natural-language requests onto the right code locations and system changes.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MiniMax M2.7: Early Echoes of Self-Evolution","https://www.minimax.io/news/minimax-m27-en",null,"2026-08-01","registry-only",null],["benchlm-svgbench","SVG-Bench","MiniMax","coding","2026","A display-only provider benchmark for generating or manipulating SVG outputs from natural-language requirements.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MiniMax M3 model card","https://huggingface.co/MiniMaxAI/MiniMax-M3",null,"2026-08-01","registry-only",null],["benchlm-swemultilingual","SWE Multilingual","MiniMax","coding","2026","A multilingual software-engineering benchmark for real-world code issue resolution across multiple programming languages.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MiniMax M2.7: Early Echoes of Self-Evolution","https://www.minimax.io/news/minimax-m27-en",null,"2026-08-01","registry-only",null],["benchlm-vibev2","VIBE V2","MiniMax","coding","2026","A display-only MiniMax provider benchmark for end-to-end coding-agent and product-building tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MiniMax M3 model card","https://huggingface.co/MiniMaxAI/MiniMax-M3",null,"2026-08-01","registry-only",null],["benchlm-vibepro","VIBE-Pro","MiniMax","coding","2026","A repo-level code generation and full-project delivery benchmark spanning web, mobile, and simulation-style implementation tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MiniMax M2.7: Early Echoes of Self-Evolution","https://www.minimax.io/news/minimax-m27-en",null,"2026-08-01","registry-only",null],["benchlm-mewc","Multi-Environment Web Challenge","MiniMax / benchmark maintainers","agents","2026","A benchmark that evaluates AI agents on multi-environment web challenges, testing navigation and task completion across diverse live web environments.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MiniMax M2.5 benchmark release surface","https://www.minimax.io/news/minimax-m25",null,"2026-08-01","registry-only",null],["benchlm-bbh","BIG-Bench Hard","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","reasoning","2022","A suite of 23 challenging tasks from the BIG-Bench collaborative benchmark where prior language models failed to exceed average human performance, even with chain-of-thought prompting.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them","https://arxiv.org/abs/2210.09261",null,"2026-08-01","registry-only",null],["benchlm-flteval","FLTEval","Mistral AI","coding","2026","A repository-level Lean 4 proof engineering benchmark that measures whether a model can complete formal proofs and correctly define new mathematical concepts inside realistic FLT project pull requests.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Leanstral: Open-Source foundation for trustworthy vibe-coding","https://mistral.ai/news/leanstral",null,"2026-08-01","registry-only",null],["benchlm-mlsbenchlite","MLS-Bench Lite","MLS-Bench","coding","2026","A 30-task subset of MLS-Bench that evaluates whether AI systems can invent generalizable and scalable machine-learning methods.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MLS-Bench","https://mls-bench.com/",null,"2026-08-01","registry-only",null],["benchlm-mmluprox","MMLU-ProX","MMLU-ProX authors","knowledge","2025","A multilingual extension of professional-level academic evaluation across many languages.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MMLU-ProX: A Multilingual Benchmark for Advanced Large Language Model Evaluation","https://arxiv.org/abs/2503.10497",null,"2026-08-01","registry-only",null],["benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","MMMU authors","multimodal","2024","A broad multimodal reasoning benchmark spanning charts, diagrams, tables, and academic visual question answering.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","https://arxiv.org/abs/2401.05508",null,"2026-08-01","registry-only",null],["mmmu","MMMU","MMMU authors","multimodal",null,"A multi-discipline multimodal understanding and reasoning benchmark for expert-level tasks.",null,"higher",null,"not-weighted",0,"active",null,null,"unknown","direct",null,null,null,null,null,"MMMU official repository","https://github.com/MMMU-Benchmark/MMMU",null,"2026-07-14","registry-only",null],["benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","MMMU-Pro authors","multimodal","2024","A harder multimodal benchmark for frontier models that combines text with images, diagrams, charts, and academic visual reasoning tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark","https://arxiv.org/abs/2409.02813",null,"2026-08-01","registry-only",null],["mmmu-pro","MMMU-Pro","MMMU-Pro authors","multimodal",null,"A more robust extension of MMMU intended to reduce shortcut-based answering.",null,"higher",null,"ranking",0.041721,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"MMMU and MMMU-Pro official repository","https://github.com/MMMU-Benchmark/MMMU",null,"2026-08-01","source-qualified",null],["benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","MMVU benchmark maintainers","multimodal","2026","A benchmark for evaluating multimodal models on video understanding tasks across multiple disciplines, emphasizing temporal reasoning and comprehension over video content.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Kimi K2.5 benchmark release surface","https://www.kimi.com/blog/kimi-k2-5.html",null,"2026-08-01","registry-only",null],["benchlm-automationbench","AutomationBench","Moonshot AI","agents","2026","An agent benchmark for completing automation workflows in reproducible task environments.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"moonshot-kimi-k3-blog","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-08-01","registry-only",null],["benchlm-babyvisionpython","BabyVision with Python","Moonshot AI","multimodal","2026","A Python-assisted BabyVision evaluation for fine-grained visual perception and grounded reasoning.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"moonshot-kimi-k3-blog","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-08-01","registry-only",null],["benchlm-deckbench","DECK-Bench (Internal)","Moonshot AI","agents","2026","Moonshot AI's internal benchmark for presentation and deck-production workflows.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"moonshot-kimi-k3-blog","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-08-01","registry-only",null],["benchlm-kimiclaw247","Kimi Claw 24/7 Bench","Moonshot AI","agents","2026","A Moonshot AI internal long-horizon agent benchmark for persistent professional coworking tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Kimi K2.7 Code","https://huggingface.co/moonshotai/Kimi-K2.7-Code",null,"2026-08-01","registry-only",null],["kimi-code-bench-v2","Kimi Code Bench v2","Moonshot AI","coding","2.0","Moonshot internal end-to-end coding tasks used in the Kimi K3 launch table; retained reference-only.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",null,"unknown","direct",null,null,"moonshot-kimi-k3-blog","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3 launch blog","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-08-01","registry-only",null],["benchlm-perceptionbench","PerceptionBench (Internal)","Moonshot AI","multimodal","2026","Moonshot AI's internal benchmark for atomic visual perception capabilities.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"moonshot-kimi-k3-blog","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-08-01","registry-only",null],["benchlm-spreadsheetbench2","SpreadsheetBench 2","Moonshot AI","agents","2026","A spreadsheet-focused benchmark for agentic analysis and editing workflows.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"moonshot-kimi-k3-blog","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-08-01","registry-only",null],["benchlm-toolathlonverified","Toolathlon-Verified","Moonshot AI","agents","2026","A verified tool-use benchmark variant for completing multi-step workflows with external tools.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"moonshot-kimi-k3-blog","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-08-01","registry-only",null],["benchlm-apexagents","APEX-Agents","Moonshot AI / APEX-Agents benchmark authors","agents","2026","A professional-services agent benchmark covering long-horizon knowledge-work tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"moonshot-kimi-k3-blog","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-08-01","registry-only",null],["benchlm-mathvisionpython","MathVision with Python","Moonshot AI / MathVision authors","multimodal","2026","A tool-augmented MathVision variant that permits Python during visual mathematics reasoning.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"moonshot-kimi-k3-blog","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-08-01","registry-only",null],["benchlm-omnidocbench","OmniDocBench","Moonshot AI / OmniDocBench authors","multimodal","2026","A document-understanding benchmark for parsing and reasoning over complex document layouts.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"moonshot-kimi-k3-blog","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-08-01","registry-only",null],["benchlm-worldvqaforceanswer","WorldVQA ForceAnswer","Moonshot AI / WorldVQA authors","multimodal","2026","A forced-answer WorldVQA variant for atomic visual world knowledge.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"moonshot-kimi-k3-blog","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-08-01","registry-only",null],["benchlm-zerobenchpython","ZeroBench_main with Python","Moonshot AI / ZeroBench authors","multimodal","2026","A Python-assisted ZeroBench_main evaluation reported as pass@5.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"moonshot-kimi-k3-blog","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-08-01","registry-only",null],["multi-swe-bench","Multi-SWE-Bench","Multi-SWE-Bench","coding",null,"Multi-language software engineering repository evaluation extending SWE-bench style tasks.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"Multi-SWE-Bench","https://benchlm.ai/benchmarks/multiSweBench",null,"2026-08-01","source-qualified",null],["musr","MuSR","MuSR authors","reasoning",null,"Multi-step soft reasoning benchmark requiring long narrative understanding and structured logical deduction.",null,"higher",null,"not-weighted",0,"archived","percent-direct-v1",1,"unknown","direct",null,null,"musr-source","production::musr-source","musr-source","MuSR","https://github.com/Zayne-sprague/MuSR",null,"2026-07-16","source-qualified",null],["benchlm-acecyberrangesolved","ACE Cyber Range Challenges Solved","NIST CAISI and UK AISI","knowledge","2026","Number of advanced cyber-range challenges solved in the joint NIST CAISI and UK AISI preliminary evaluation.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"UK AISI and CAISI Preliminary Assessment of Kimi K3's Cyber Capabilities","https://www.nist.gov/news-events/news/2026/07/uk-aisi-caisi-preliminary-assessment-kimi-k3s-cyber-capabilities",null,"2026-08-01","registry-only",null],["benchlm-lastonescyberrangesteps","The Last Ones Average Progress","NIST CAISI and UK AISI","knowledge","2026","Average step reached on a 32-step long-horizon cyber range.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"UK AISI and CAISI Preliminary Assessment of Kimi K3's Cyber Capabilities","https://www.nist.gov/news-events/news/2026/07/uk-aisi-caisi-preliminary-assessment-kimi-k3s-cyber-capabilities",null,"2026-08-01","registry-only",null],["benchlm-lastonescyberrangecompletion","The Last Ones Cyber Range Completion Rate","NIST CAISI and UK AISI","knowledge","2026","Share of runs that completed the 32-step cyber range within the 100-million-token limit.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"UK AISI and CAISI Preliminary Assessment of Kimi K3's Cyber Capabilities","https://www.nist.gov/news-events/news/2026/07/uk-aisi-caisi-preliminary-assessment-kimi-k3s-cyber-capabilities",null,"2026-08-01","registry-only",null],["ocrbench-v2","OCRBench v2","OCRBench authors","multimodal","2","Multimodal OCR and document understanding benchmark covering text recognition in complex real-world images.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"unknown","direct",null,null,"ocrbench-v2-source","production::ocrbench-v2-source","ocrbench-v2-source","OCRBench","https://github.com/Yuliang-Liu/MultimodalOCR",null,"2026-08-01","source-qualified",null],["officeqa-pro","OfficeQA Pro","OfficeQA authors","agents",null,"Document and spreadsheet workplace QA benchmark measuring office-style multi-file reasoning and calculation.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"unknown","direct",null,null,"officeqa-pro-source","production::officeqa-pro-source","officeqa-pro-source","OfficeQA official repository","https://github.com/databricks/officeqa",null,"2026-08-01","source-qualified",null],["browsecomp","BrowseComp","OpenAI","research",null,"A benchmark for browsing agents that must locate difficult-to-find information on the web.",null,"higher",null,"ranking",0.016366,"active","percent-direct-v1",1,"low","direct",null,null,null,null,null,"BrowseComp","https://openai.com/index/browsecomp/",null,"2026-08-01","source-qualified",null],["benchlm-frontierscience","FrontierScience","OpenAI","knowledge","2026","A benchmark for research-level scientific reasoning, designed to separate frontier models on difficult science tasks that mix domain knowledge with deep reasoning.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"FrontierScience","https://openai.com/index/frontierscience/",null,"2026-08-01","registry-only",null],["benchlm-genebenchpro","GeneBench-Pro","OpenAI","reasoning","2026","A multistage statistical-reasoning benchmark for genomics and biological-data analysis agents.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GeneBench-Pro: Evaluating Multistage Statistical Reasoning","https://cdn.openai.com/pdf/21938268-21af-442f-af93-3b2249afb241/genebench-pro.pdf",null,"2026-08-01","registry-only",null],["benchlm-graphwalksbfs128k","Graphwalks BFS 0K-128K","OpenAI","reasoning","2026","Long-context graph traversal benchmark using breadth-first search tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Introducing GPT-5.4 mini and nano","https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",null,"2026-08-01","registry-only",null],["benchlm-graphwalksparents128k","Graphwalks parents 0-128K","OpenAI","reasoning","2026","Long-context benchmark for recovering parent relationships inside graph tasks.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Introducing GPT-5.4 mini and nano","https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",null,"2026-08-01","registry-only",null],["healthbench-hard","HealthBench Hard","OpenAI","research",null,"OpenAI HealthBench hard subset evaluating model helpfulness and safety on challenging healthcare conversations.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"unknown","direct",null,null,"openai-healthbench","production::openai-healthbench","openai-healthbench","HealthBench","https://openai.com/index/healthbench/",null,"2026-08-01","source-qualified",null],["benchlm-hlenotools","Humanity's Last Exam without tools","OpenAI","knowledge","2026","Tool-free variant of Humanity's Last Exam that isolates a model's raw frontier reasoning.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Introducing GPT-5.4 mini and nano","https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",null,"2026-08-01","registry-only",null],["mle-bench","MLE-Bench","OpenAI","coding","Partial-30","Machine-learning engineering benchmark where agents compete on Kaggle-style tasks; Google reports Average Position Score on the Partial-30 subset.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"low","direct",null,null,"openai-mle-bench","production::openai-mle-bench","openai-mle-bench","MLE-Bench repository","https://github.com/openai/mle-bench",null,"2026-07-21","source-qualified",null],["benchlm-mmmlu","MMMLU","OpenAI","knowledge","2026","A multilingual MMLU-style benchmark reported in provider evaluation tables.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MMMLU","https://huggingface.co/datasets/openai/MMMLU",null,"2026-08-01","registry-only",null],["benchlm-mmmupropython","MMMU-Pro with Python","OpenAI","multimodal","2026","Tool-augmented MMMU-Pro variant that allows Python assistance during multimodal reasoning.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Introducing GPT-5.4 mini and nano","https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",null,"2026-08-01","registry-only",null],["mrcr-v2","MRCRv2","OpenAI","research","2","Multi-needle long-context retrieval and reasoning benchmark stressing multi-document evidence gathering.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"low","direct",null,null,"openai-mrcr","production::openai-mrcr","openai-mrcr","MRCR","https://huggingface.co/datasets/openai/mrcr",null,"2026-08-01","source-qualified",null],["benchlm-omnidocbench15","OmniDocBench 1.5","OpenAI","multimodal","2026","A document understanding benchmark used in frontier-model comparison tables to measure extraction and grounded reasoning quality on complex documents.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Introducing GPT-5.4 mini and nano","https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",null,"2026-08-01","registry-only",null],["benchlm-mrcrv2-128-256","OpenAI MRCR v2 8-needle 128K-256K","OpenAI","reasoning","2026","MRCR v2 slice focused on very long contexts at 128K-256K lengths.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Introducing GPT-5.4 mini and nano","https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",null,"2026-08-01","registry-only",null],["benchlm-mrcrv2-64-128","OpenAI MRCR v2 8-needle 64K-128K","OpenAI","reasoning","2026","MRCR v2 slice focused on long-context retrieval at 64K-128K lengths.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Introducing GPT-5.4 mini and nano","https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",null,"2026-08-01","registry-only",null],["paperbench","PaperBench","OpenAI","research",null,"An evaluation of agents attempting to replicate machine-learning research from published papers.",null,"higher",null,"ranking",0.034566,"active","percent-direct-v1",1,"low","direct",null,null,null,null,null,"PaperBench","https://openai.com/index/paperbench/",null,"2026-07-15","source-qualified",null],["simpleqa","SimpleQA","OpenAI","research",null,"A short-answer factuality benchmark designed to measure correctness on fact-seeking questions.",null,"higher",null,"ranking",0.034566,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"Introducing SimpleQA","https://openai.com/index/introducing-simpleqa/",null,"2026-07-15","source-qualified",null],["swe-lancer","SWE-Lancer","OpenAI","coding",null,"Freelance-style software engineering benchmark mapping model performance to real paid engineering task outcomes.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"low","direct",null,null,"registry-swe-lancer","production::registry-swe-lancer","registry-swe-lancer","SWE-Lancer","https://openai.com/index/swe-lancer/",null,"2026-07-21","registry-only",null],["simpleqa-verified","SimpleQA Verified","OpenAI / Google","research","verified","Factual short-answer evaluation derived from SimpleQA with stricter verification protocols against known ground-truth answers.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"medium","direct",null,null,null,null,null,"SimpleQA","https://openai.com/index/introducing-simpleqa/",null,"2026-07-21","registry-only",null],["math-500","MATH-500","OpenAI / Hendrycks","mathematics",null,"A 500-problem competition-math subset from the MATH dataset spanning algebra, geometry, number theory, and related domains.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"medium","direct",null,null,"aa-math-500","production::aa-math-500","aa-math-500","MATH-500 Leaderboard","https://artificialanalysis.ai/evaluations/math-500","2026-07-15","2026-07-16","source-qualified",null],["benchlm-openhandsindex","OpenHands Index","OpenHands","agents","2025","A holistic coding-agent benchmark that evaluates AI agents across issue resolution, frontend work, greenfield development, testing, and information gathering.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"OpenHands Index methodology","https://index.openhands.dev/about",null,"2026-08-01","registry-only",null],["osworld-verified","OSWorld-Verified","OSWorld","agents","Verified","Evaluates computer-use agents in verified desktop interaction tasks.",null,"higher",null,"ranking",0.016366,"active","percent-direct-v1",1,"unknown","direct",null,null,"refresh-osworld-v1","production::refresh-osworld-v1","refresh-osworld-v1","OSWorld-Verified","https://os-world.github.io/",null,"2026-08-01","source-qualified",null],["osworld","OSWorld","OSWorld authors","agents",null,"A benchmark for multimodal agents performing tasks in real computer operating systems.",null,"higher",null,"ranking",0.016366,"rolling","percent-direct-v1",1,"low","direct",null,null,null,null,null,"OSWorld official repository","https://github.com/xlang-ai/OSWorld",null,"2026-08-01","source-qualified",null],["benchlm-bullshitbenchv2","BullshitBench v2","Peter Gostev","reasoning","2025","A benchmark that tests whether AI models challenge nonsensical, ill-posed, or logically flawed prompts instead of confidently generating incorrect answers. Measures the critical ability to push back on bad input.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"BullshitBench: Measuring whether AI models challenge nonsensical prompts","https://petergpt.github.io/bullshit-benchmark/",null,"2026-08-01","registry-only",null],["posttrain-bench","PostTrainBench","PostTrainBench","coding","public","Autonomous post-training of small language models under fixed compute and time budgets.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"unknown","direct",null,null,"posttrain-bench-site","production::posttrain-bench-site","posttrain-bench-site","PostTrainBench","https://posttrainbench.com/",null,"2026-08-01","source-qualified",null],["program-bench","ProgramBench","ProgramBench","coding","public","Rebuild command-line programs from binary behaviour and documentation without source code.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"medium","direct",null,null,"program-bench-site","production::program-bench-site","program-bench-site","ProgramBench","https://www.vals.ai/benchmarks/programbench",null,"2026-07-21","source-qualified",null],["benchlm-aineedle","AI-Needle","Qwen","reasoning","2026","A long-context retrieval benchmark that measures whether a model can recover relevant information embedded deep inside very long contexts.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-ai2dtest","AI2D test split","Qwen","multimodal","2026","A diagram understanding benchmark focused on scientific and educational visual question answering.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-ccocr","CC-OCR","Qwen","multimodal","2026","An OCR-focused benchmark for reading and extracting text from visually complex documents and images.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-chatcvqa","ChatCVQA","Qwen","multimodal","2026","A conversational visual QA benchmark that tests multi-turn grounded answering over images and documents.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-countbench","CountBench","Qwen","multimodal","2026","A visual counting benchmark that tests whether a model can count objects and entities reliably in complex scenes.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["cowork-bench","CoWorkBench","Qwen","agents","2026-08","Qwen in-house long-horizon office-work benchmark covering computer science, finance, law, medical and other productivity domains.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"unknown","direct",null,null,"qwen-3-8-27b-huggingface-2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B model card","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-14","2026-08-15","source-qualified",null],["benchlm-dynamath","DynaMath","Qwen","multimodal","2026","A multimodal benchmark for dynamic mathematical reasoning over visual and structured inputs.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-erqa","ERQA","Qwen","multimodal","2026","A grounded visual reasoning benchmark focused on evidence-based question answering over real images.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","Qwen","mathematics","2026","A February 2026 HMMT slice used in newer frontier-model math comparisons.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","Qwen","mathematics","2025","A November 2025 HMMT slice for high-end mathematical reasoning comparisons.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-include","INCLUDE","Qwen","knowledge","2026","A multilingual benchmark used in provider tables to measure inclusive language coverage and cross-lingual understanding beyond common high-resource languages.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-mathvision","MathVision","Qwen","multimodal","2026","A visual mathematics benchmark that tests whether a model can solve math problems grounded in diagrams, equations, figures, and other visual inputs.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-maxife","MAXIFE","Qwen","knowledge","2026","A multilingual instruction-following and understanding benchmark row published in Qwen's launch comparisons.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-mcptasks","MCP-Tasks","Qwen","agents","2026","A Model Context Protocol task benchmark used in Qwen's launch tables to measure practical execution over MCP-style tools and integrations.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-mlvuavg","MLVU mean average","Qwen","multimodal","2026","A multi-task video understanding benchmark averaged across MLVU categories.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-mmanswerbench","MMAnswerBench","Qwen","mathematics","2026","A multimodal mathematical reasoning benchmark that tests whether models can answer visually grounded math questions correctly.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-mmlongbenchdoc","MMLongBench-Doc","Qwen","multimodal","2026","A long-document multimodal benchmark for grounded reasoning over extended document contexts.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-mmluredux","MMLU-Redux","Qwen","knowledge","2026","A harder refresh of MMLU intended to keep broad knowledge evaluation useful after the original benchmark became too easy for frontier models.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-mstar","MStar","Qwen","multimodal","2026","A general visual question-answering benchmark used in provider tables for real-image reasoning quality.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-nova63","NOVA-63","Qwen","knowledge","2026","A broad multilingual benchmark row from Qwen's launch comparisons intended to measure cross-lingual capability beyond a single language family.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-odinw13","ODINW13","Qwen","multimodal","2026","A visual detection and grounding benchmark slice used to compare zero-shot object understanding across diverse domains.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-polymath","PolyMath","Qwen","knowledge","2026","A multilingual mathematical reasoning benchmark that tests whether math performance transfers across languages rather than only in English.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-qwenclawbench","QwenClawBench","Qwen","agents","2026","Qwen's internal OpenClaw-style benchmark for measuring broad real-world agent performance across practical productivity and research tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["qwen-swe-bench","QwenSWEBench","Qwen","coding","2026-08","Qwen in-house software-engineering benchmark evaluated with the Claude Code harness and published on the Qwen3.8-27B card.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"unknown","direct",null,null,"qwen-3-8-27b-huggingface-2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B model card","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-14","2026-08-15","source-qualified",null],["benchlm-qwenwebbench","QwenWebBench","Qwen","agents","2026","A Qwen benchmark for artifact and webpage generation quality reported as an Elo-style rating.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-realworldqa","RealWorldQA","Qwen","multimodal","2026","A grounded visual QA benchmark focused on answering practical questions about real-world images and scenes.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["recreation-bench","RecreationBench","Qwen","agents","2026-08","Qwen in-house application-recreation benchmark for hybrid agents across desktop, mobile and web platforms.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"unknown","direct",null,null,"qwen-3-8-27b-huggingface-2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B model card","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-14","2026-08-15","source-qualified",null],["benchlm-tirbench","TIR-Bench","Qwen","multimodal","2026","A visual agent benchmark for interface reasoning and task execution over screenshots or software surfaces.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-videommewithsub","Video-MME with subtitle","Qwen","multimodal","2026","A video understanding benchmark that allows subtitle access when answering multimodal questions about videos.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-videommenosub","Video-MME without subtitle","Qwen","multimodal","2026","A stricter Video-MME setting that removes subtitle help and tests video understanding from visual and audio context alone.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-vwt2klite","VWT2k-lite","Qwen","knowledge","2026","A lighter multilingual benchmark slice published in provider tables for broad cross-lingual transfer and understanding.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-wemath","We-Math","Qwen","multimodal","2026","A multimodal math benchmark for visually grounded mathematical reasoning and answer generation.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-wideresearch","WideResearch","Qwen","agents","2026","A broad research-agent benchmark for open-ended information gathering, synthesis, and answer construction across wide search spaces.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Qwen3.6 launch benchmarks","https://qwen.ai/blog?id=qwen3.6",null,"2026-08-01","registry-only",null],["benchlm-healthbenchprofessional","HealthBench Professional","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","knowledge","2026","An open benchmark for clinician-facing model responses across care consult, writing and documentation, and medical research tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"HealthBench Professional: Evaluating Large Language Models on Real Clinician Chats","https://arxiv.org/abs/2604.27470",null,"2026-08-01","registry-only",null],["benchlm-refcocoavg","RefCOCO average","RefCOCO dataset authors","multimodal","2026","A referring-expression grounding benchmark averaged across RefCOCO variants to test whether a model can localize described objects correctly.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"RefCOCO referring expression datasets","https://github.com/lichengunc/refer",null,"2026-08-01","registry-only",null],["benchlm-frontierbench","FrontierBench v0.1","Ryan Marten, Alex Shaw, Andy Konwinski, Harbor, and the Laude Institute","agents","2026","A continuously maintained professional computer-work benchmark from the team behind Terminal-Bench and Harbor.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"FrontierBench","https://www.frontierbench.ai/",null,"2026-08-01","registry-only",null],["benchlm-ctirealm","CTI-REALM","Sakana AI","agents","2026","A cybersecurity benchmark that measures whether an agent can turn raw threat-intelligence reports into working detection rules.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Introducing Fugu-Cyber","https://sakana.ai/fugu-cyber-release/",null,"2026-08-01","registry-only",null],["benchlm-sweatlasrefactoring","SWE-Atlas Refactoring","Scale AI","agents","2026","A Scale SWE-Atlas software-engineering agent benchmark focused on refactoring tasks.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"SWE-Atlas","https://labs.scale.com/papers/sweatlas",null,"2026-08-01","registry-only",null],["swe-bench-pro","SWE-Bench Pro","Scale AI","coding","v1","Public single-attempt software engineering tasks drawn from repositories outside the original SWE-bench set.",null,"higher",null,"ranking",0.031694,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"SWE-Bench Pro","https://scale.com/leaderboard/swe_bench_pro_public",null,"2026-08-01","source-qualified",null],["scicode","SciCode","SciCode authors","coding",null,"A benchmark for generating scientific-computing code from expert-authored specifications.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"low","direct",null,null,null,null,null,"SciCode official repository","https://github.com/scicode-bench/SciCode",null,"2026-08-01","source-qualified",null],["screenspot-pro","ScreenSpot-Pro","ScreenSpot","multimodal","Pro","GUI grounding benchmark measuring precise visual localization of UI elements for computer-use agents.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"ScreenSpot-Pro (referenced in Gemini evaluations)","https://github.com/njucckevin/SeeClick",null,"2026-08-01","registry-only",null],["benchlm-exploitbench","ExploitBench v8-bench","Seunghyun Lee, David Brumley, Carnegie Mellon University","knowledge","2026","A cybersecurity benchmark for evaluating LLM agents on full-control V8 exploit synthesis using 16 measured exploit capability flags.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"ExploitBench","https://exploitbench.ai/",null,"2026-08-01","registry-only",null],["benchlm-taubench","Tool-Agent-User Benchmark","Shunyu Yao, Noah Shinn, Pedram Razavi, Karthik Narasimhan","agents","2024","Original TAU-bench evaluates a model-driven agent in simulated airline and retail customer-service conversations with domain tools, database state, and policy constraints.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"τ-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","https://arxiv.org/abs/2406.12045",null,"2026-08-01","registry-only",null],["benchlm-webarena","WebArena Web Agent Benchmark","Shuyan Zhou, Frank F. Xu, Hao Zhu, Xuhui Zhou, Robert Lo, Abishek Sridhar, Xianyi Cheng, Tianyue Ou, Yonatan Bisk, Daniel Fried, Uri Alon, Graham Neubig","agents","2024","WebArena tests whether a browser-agent system can complete 812 long-horizon tasks inside self-hosted replicas of functional websites. It checks the requested end state, so a result reflects the model, agent scaffold, browser interface, action budget, and evaluator together—not the base model alone.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"WebArena: A Realistic Web Environment for Building Autonomous Agents","https://arxiv.org/abs/2307.13854",null,"2026-08-01","registry-only",null],["tau2-bench","τ²-Bench Telecom","Sierra / Artificial Analysis","agents","2","Dual-control conversational agent benchmark for telecom support where agent and user must coordinate tool actions.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"low","direct",null,null,"aa-tau2-bench","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Leaderboard","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-08-01","source-qualified",null],["tau3-banking","τ³-Banking","Sierra / Artificial Analysis","agents","3","Fintech customer-support agent benchmark requiring knowledge-base navigation and multi-step tool use on banking workflows.",null,"higher",null,"ranking",0.011925,"active","percent-direct-v1",1,"low","direct",null,null,"aa-evaluation-tau3-banking-2026-08-27","production::aa-evaluation-tau3-banking-2026-08-27","aa-evaluation-tau3-banking-2026-08-27","τ³-Banking Leaderboard","https://artificialanalysis.ai/evaluations/tau3-banking",null,"2026-07-15","source-qualified",null],["tau-bench","τ-bench","Sierra Research","agents",null,"A benchmark for tool-using language agents interacting with simulated user and business environments.",null,"higher",null,"ranking",0.016366,"active","percent-direct-v1",1,"unknown","direct",null,null,"refresh-tau-bench-legacy","production::refresh-tau-bench-legacy","refresh-tau-bench-legacy","tau-bench official repository","https://github.com/sierra-research/tau-bench",null,"2026-07-15","source-qualified",null],["benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","Sierra Research","agents","2026","τ³-bench is the current evolution of Sierra's tool-agent-user framework, adding corrected task releases and newer knowledge and voice evaluation modes alongside airline, retail, and telecom.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,"refresh-tau2-bench-repo","production::refresh-tau2-bench-repo","refresh-tau2-bench-repo","Official τ³-bench repository and release notes","https://github.com/sierra-research/tau2-bench",null,"2026-08-01","registry-only",null],["benchlm-gmmlu","Global MMLU","Singh et al.","knowledge","2024","MMLU-style knowledge evaluation across 42 high- and low-resource languages.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Global MMLU: Understanding and addressing cultural and linguistic biases in multilingual evaluation","https://arxiv.org/abs/2412.03304",null,"2026-08-01","registry-only",null],["benchlm-seniorswebench","Senior SWE-Bench","Snorkel AI","coding","2026","A Snorkel AI benchmark of senior-level software engineering tasks emphasizing under-specified feature work, bug/performance investigation, and taste-based correctness.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Senior SWE-Bench","https://senior-swe-bench.snorkel.ai/",null,"2026-08-01","registry-only",null],["spider2","Spider 2.0","Spider","coding","2.0","Enterprise text-to-SQL benchmark spanning complex databases and multi-dialect SQL generation for realistic analytics workloads.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"medium","direct",null,null,"registry-spider2","production::registry-spider2","registry-spider2","Spider 2.0","https://spider2-sql.github.io/",null,"2026-07-21","registry-only",null],["benchlm-spider2lite","Spider 2.0-Lite","Spider 2.0 authors","coding","2024","A text-to-SQL benchmark over realistic warehouse-scale schemas, reported by Interfaze for model comparison.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Spider 2.0: Evaluating Language Models on Real-World Enterprise Text-to-SQL Workflows","https://github.com/xlang-ai/Spider2",null,"2026-08-01","registry-only",null],["cybench","Cybench","Stanford / Cybench authors","agents",null,"Professional-level Capture-the-Flag cybersecurity agent benchmark measuring unguided end-to-end task solve rates.",null,"higher",null,"ranking",0.010221,"active","percent-direct-v1",1,"low","direct",null,null,"cybench-official","production::cybench-official","cybench-official","Cybench leaderboard","https://cybench.github.io/","2026-07-14","2026-08-01","source-qualified",null],["benchlm-truthfulqa","TruthfulQA","Stephanie Lin, Jacob Hilton, Owain Evans","knowledge","2021","A benchmark designed to measure whether language models produce truthful answers instead of repeating common misconceptions or misleading falsehoods.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"TruthfulQA: Measuring How Models Mimic Human Falsehoods","https://arxiv.org/abs/2109.07958",null,"2026-08-01","registry-only",null],["benchlm-gbaeval","GBA-Eval","Stephen Yang","knowledge","2026","An agentic coding benchmark that asks models to build a Game Boy Advance emulator from scratch and grades emulator behavior against procedural, audio, and gameplay tests.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GBA-Eval","https://gbaeval.com/",null,"2026-08-01","registry-only",null],["super-gpqa","SuperGPQA","SuperGPQA authors","reasoning",null,"Graduate-level Google-proof Q&A expansion beyond GPQA Diamond for harder knowledge-reasoning items.",null,"higher",null,"ranking",0.025006,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"SuperGPQA","https://benchlm.ai/benchmarks/superGpqa",null,"2026-07-15","source-qualified",null],["benchlm-chartographywithtools","Chartography with image and code tools","Surge AI and Anthropic","multimodal","2026","Professional chart understanding with a container, standard libraries, and image cropping.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Chartography","https://surgehq.ai/blog/chartography",null,"2026-08-01","registry-only",null],["benchlm-chartography","Chartography without tools","Surge AI and Anthropic","multimodal","2026","Professional chart understanding across 100 specialized chart types with expert-set answer tolerances.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Chartography","https://surgehq.ai/blog/chartography",null,"2026-08-01","registry-only",null],["benchlm-gdppdfwithtools","GDP.pdf mean criteria pass rate with tools","Surge AI and Anthropic","multimodal","2026","Professional document understanding with a container, standard libraries, and image cropping.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GDP.pdf","https://surgehq.ai/blog/gdp-pdf-can-100b-ai-models-master-the-documents-that-run-the-world",null,"2026-08-01","registry-only",null],["benchlm-gdppdf","GDP.pdf mean criteria pass rate without tools","Surge AI and Anthropic","multimodal","2026","Professional document understanding over 100 real-world PDFs from ten domains.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GDP.pdf","https://surgehq.ai/blog/gdp-pdf-can-100b-ai-models-master-the-documents-that-run-the-world",null,"2026-08-01","registry-only",null],["benchlm-riemannbenchwithtools","RiemannBench with tools","Surge AI and Anthropic","mathematics","2026","Research-level mathematics problems with unique programmatically verified closed-form answers and tool access.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["benchlm-riemannbench","RiemannBench without tools","Surge AI and Anthropic","mathematics","2026","Research-level mathematics problems with unique programmatically verified closed-form answers.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Claude Opus 5 System Card","https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",null,"2026-08-01","registry-only",null],["swe-bench-verified","SWE-bench Verified","SWE-bench authors","coding","Verified","A human-validated subset of real GitHub software-engineering issue resolution tasks.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"medium","direct",null,null,"refresh-swe-bench-site","production::refresh-swe-bench-site","refresh-swe-bench-site","SWE-bench","https://www.swebench.com/",null,"2026-08-01","source-qualified",null],["benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","SWE-bench team","knowledge","2025","A multilingual extension of SWE-bench covering 300 problems across 9 programming languages, testing code generation and bug fixing beyond Python.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"SWE-bench Multilingual","https://www.swebench.com/multilingual",null,"2026-08-01","registry-only",null],["benchlm-swemultimodal","SWE-bench Multimodal","SWE-bench team","coding","2025","A multimodal variant of SWE-bench that adds visual context such as screenshots and design mockups to software engineering issue descriptions.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"SWE-bench Multimodal","https://www.swebench.com/multimodal",null,"2026-08-01","registry-only",null],["swe-rebench","SWE-Rebench","SWE-Rebench","coding",null,"Software engineering repository tasks designed as a refreshed SWE-bench style evaluation.",null,"higher",null,"not-weighted",0,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"SWE-Rebench","https://benchlm.ai/benchmarks/sweRebench",null,"2026-08-01","source-qualified",null],["terminal-bench-2","Terminal-Bench 2.0","Terminal-Bench contributors","coding","2026","A benchmark for agentic software engineering tasks executed in real terminal environments. DeepSeek reports it in the agentic section, while BenchLM also mirrors it in coding for models that publish it as a developer-task signal.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Terminal-Bench 2.0","https://www.tbench.ai/",null,"2026-08-01","registry-only",null],["terminal-bench","Terminal-Bench","Terminal-Bench team","agents","2.1","A benchmark for agents performing practical tasks in reproducible terminal environments.",null,"higher",null,"ranking",0.016366,"rolling","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"Terminal-Bench 2.1","https://www.tbench.ai/news/terminal-bench-2-1",null,"2026-07-27","source-qualified",null],["the-agent-company","TheAgentCompany","TheAgentCompany","agents",null,"Realistic office-work agent evaluation where models complete multi-day digital workplace tasks using browsers, docs, and internal tools.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"low","direct",null,null,"registry-the-agent-company","production::registry-the-agent-company","registry-the-agent-company","TheAgentCompany","https://the-agent-company.com/",null,"2026-07-21","registry-only",null],["agentbench","AgentBench","THUDM","agents",null,"A multi-environment benchmark for evaluating language models as interactive agents.",null,"higher",null,"not-weighted",0,"archived",null,null,"unknown","direct",null,null,null,null,null,"AgentBench official repository","https://github.com/THUDM/AgentBench",null,"2026-07-14","registry-only",null],["longbench-v2","LongBench v2","THUDM","long-context","2","A long-context reasoning benchmark built around demanding real-world tasks.",null,"higher",null,"not-weighted",0,"active",null,null,"unknown","direct",null,null,null,null,null,"LongBench v2 official repository","https://github.com/THUDM/LongBench",null,"2026-08-01","registry-only",null],["mmlu-pro","MMLU-Pro","TIGER-Lab","research",null,"A more challenging multiple-choice knowledge and reasoning benchmark derived from MMLU.",null,"higher",null,"ranking",0.034569,"active","percent-direct-v1",1,"unknown","direct",null,null,null,null,null,"MMLU-Pro official repository","https://github.com/TIGER-AI-Lab/MMLU-Pro",null,"2026-07-15","source-qualified",null],["benchlm-openbookqa","OpenBookQA","Todor Mihaylov, Peter Clark, Tushar Khot, Ashish Sabharwal","knowledge","2018","A science question-answering benchmark that tests whether models can apply a small open-book set of elementary science facts to multi-step reasoning questions.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Can a Suit of Armor Conduct Electricity? A New Dataset for Open Book Question Answering","https://arxiv.org/abs/1809.02789",null,"2026-08-01","registry-only",null],["benchlm-tolokaarena","Toloka Arena","Toloka","knowledge","2026","An independent agentic-intelligence evaluation from Toloka using private simulated workflows and a pass^5 metric.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Toloka Arena","https://toloka.ai/arena",null,"2026-08-01","registry-only",null],["benchlm-alebench","Agents Last Exam","UC Berkeley RDI","knowledge","2026","A benchmark for agentic professional workflows with verifiable success criteria, reporting pass rates and partial scores for model plus agent-harness rows.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Agents Last Exam","https://agents-last-exam.org/leaderboard",null,"2026-08-01","registry-only",null],["cybergym","CyberGym","UC Berkeley SunBlaze","agents",null,"Cybersecurity agent benchmark focused on finding and validating software vulnerabilities under constrained tool environments.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"low","direct",null,null,"cybergym-official","production::cybergym-official","cybergym-official","CyberGym repository","https://github.com/sunblaze-ucb/cybergym",null,"2026-08-01","registry-only",null],["factscore","FActScore","University of Washington","research",null,"Fine-grained atomic factuality scoring for long-form generation against retrieved evidence units.",null,"higher",null,"reference",0,"archived","percent-direct-v1",1,"medium","direct",null,null,"registry-factscore","production::registry-factscore","registry-factscore","FActScore","https://github.com/shmsw25/FActScore",null,"2026-07-21","registry-only",null],["benchlm-pokeragent","Vals Agent Poker Bench","Vals AI","knowledge","2026","Which model can make the most money playing poker?",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Agent Poker Bench","https://www.vals.ai/benchmarks/poker_agent",null,"2026-08-01","registry-only",null],["benchlm-valsaime","Vals AIME","Vals AI","knowledge","2026","Challenging national math exam given to top high-school students",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"AIME","https://www.vals.ai/benchmarks/aime",null,"2026-08-01","registry-only",null],["benchlm-valscaselawv2","Vals CaseLaw v2","Vals AI","knowledge","2026","Vals AI private question-answer benchmark over Canadian court cases.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"CaseLaw v2","https://www.vals.ai/benchmarks/case_law_v2",null,"2026-08-01","registry-only",null],["benchlm-codemigration","Vals Code Migration","Vals AI","knowledge","2026","Can language models reimplement real-world programs in another language?",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Code Migration","https://www.vals.ai/benchmarks/code-migration",null,"2026-08-01","registry-only",null],["benchlm-valscorpfinv2","Vals CorpFin v2","Vals AI","knowledge","2026","Vals AI private benchmark for understanding long-context credit agreements.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"CorpFin v2","https://www.vals.ai/benchmarks/corp_fin_v2",null,"2026-08-01","registry-only",null],["benchlm-cyber","Vals CyberBench","Vals AI","knowledge","2026","Can autonomous agents craft PoC inputs that trigger OSS-Fuzz vulnerabilities—and stop crashing after the fix?",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"CyberBench","https://www.vals.ai/benchmarks/cyber",null,"2026-08-01","registry-only",null],["benchlm-emb","Vals EMB","Vals AI","knowledge","2026","Evaluating agents on Excel-based financial modeling tasks",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"EMB","https://www.vals.ai/benchmarks/emb",null,"2026-08-01","registry-only",null],["benchlm-hlab","Vals Harvey's Legal Agent Benchmark","Vals AI","knowledge","2026","Tests an agent's ability to complete legal work using documents, spreadsheets, presentations, and file-system tools",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Harvey's Legal Agent Benchmark","https://www.vals.ai/benchmarks/hlab",null,"2026-08-01","registry-only",null],["benchlm-valsindex","Vals Index v1.2","Vals AI","knowledge","2026","Vals AI composite benchmark across finance and coding tasks, including Finance Agent v2, CorpFin v2, SWE-bench, Terminal-Bench 2.1, and Vibe Code Bench.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Vals Index","https://www.vals.ai/benchmarks/vals_index",null,"2026-08-01","registry-only",null],["benchlm-valsioi","Vals IOI","Vals AI","knowledge","2026","Based on the International Olympiad in Informatics",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"IOI","https://www.vals.ai/benchmarks/ioi",null,"2026-08-01","registry-only",null],["benchlm-legalresearchbench","Vals Legal Research Bench","Vals AI","knowledge","2026","Evaluating agents on legal research tasks across diverse areas of US law",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Legal Research Bench","https://www.vals.ai/benchmarks/legal_research",null,"2026-08-01","registry-only",null],["benchlm-valslegalbench","Vals LegalBench","Vals AI","knowledge","2026","Vals AI legal benchmark with issue, rule, conclusion, interpretation, and rhetoric task views.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"LegalBench","https://www.vals.ai/benchmarks/legal_bench",null,"2026-08-01","registry-only",null],["benchlm-valsmath500","Vals MATH 500","Vals AI","knowledge","2026","Academic math benchmark on probability, algebra, and trigonometry",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MATH 500","https://www.vals.ai/benchmarks/math500",null,"2026-08-01","registry-only",null],["benchlm-valsmedcode","Vals MedCode","Vals AI","knowledge","2026","Vals AI healthcare benchmark for whether models can support the medical billing process.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MedCode","https://www.vals.ai/benchmarks/medcode",null,"2026-08-01","registry-only",null],["benchlm-valsmedqa","Vals MedQA","Vals AI","knowledge","2026","Evaluating language model bias in medical questions.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MedQA","https://www.vals.ai/benchmarks/medqa",null,"2026-08-01","registry-only",null],["benchlm-valsmedscribe","Vals MedScribe","Vals AI","knowledge","2026","Vals AI healthcare benchmark for whether models can support doctors with administrative work.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MedScribe","https://www.vals.ai/benchmarks/medscribe",null,"2026-08-01","registry-only",null],["benchlm-valsmgsm","Vals MGSM","Vals AI","knowledge","2026","A multilingual benchmark for mathematical questions.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MGSM","https://www.vals.ai/benchmarks/mgsm",null,"2026-08-01","registry-only",null],["benchlm-valsmmmu","Vals MMMU","Vals AI","knowledge","2026","Multimodal Multi-task Benchmark",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MMMU","https://www.vals.ai/benchmarks/mmmu",null,"2026-08-01","registry-only",null],["benchlm-valsmortgagetax","Vals MortgageTax","Vals AI","knowledge","2026","Vals AI benchmark for mortgage and tax document reasoning, including semantic and numerical extraction task views.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MortgageTax","https://www.vals.ai/benchmarks/mortgage_tax",null,"2026-08-01","registry-only",null],["benchlm-valsmultimodalindex","Vals Multimodal Index v1.1","Vals AI","knowledge","2026","Vals AI multimodal composite across finance, coding, education, and mortgage-tax task families.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Vals Multimodal Index","https://www.vals.ai/benchmarks/vals_multimodal_index",null,"2026-08-01","registry-only",null],["benchlm-valsprogrambench","Vals ProgramBench","Vals AI","knowledge","2026","Can language models rebuild programs from scratch?",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,"program-bench-site","production::program-bench-site","program-bench-site","ProgramBench","https://www.vals.ai/benchmarks/programbench",null,"2026-08-01","registry-only",null],["benchlm-valsproofbench","Vals ProofBench","Vals AI","knowledge","2026","Vals AI automated theorem-proving benchmark.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"ProofBench","https://www.vals.ai/benchmarks/proof_bench",null,"2026-08-01","registry-only",null],["benchlm-publicbenefitsbenchv1","Vals Public Benefits Bench v1","Vals AI","knowledge","2026","Can AI help people navigate SNAP benefits?",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Public Benefits Bench v1","https://www.vals.ai/benchmarks/public-benefits-bench-v1",null,"2026-08-01","registry-only",null],["benchlm-publicbenefitsbench","Vals Public Benefits Bench v1.1","Vals AI","knowledge","2026","Can AI help people navigate SNAP benefits?",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Public Benefits Bench v1.1","https://www.vals.ai/benchmarks/public-benefits-bench",null,"2026-08-01","registry-only",null],["benchlm-sage","Vals SAGE","Vals AI","knowledge","2026","Student Assessment with Generative Evaluation",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"SAGE","https://www.vals.ai/benchmarks/sage",null,"2026-08-01","registry-only",null],["benchlm-skillsbench","Vals SkillsBench","Vals AI","knowledge","2026","How important are skills for agents?",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"SkillsBench","https://www.vals.ai/benchmarks/skillsbench",null,"2026-08-01","registry-only",null],["benchlm-taxevalv2","Vals TaxEval v2","Vals AI","knowledge","2026","A Vals-created set of questions and responses to tax questions",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"TaxEval v2","https://www.vals.ai/benchmarks/tax_eval_v2",null,"2026-08-01","registry-only",null],["benchlm-valsterminalbench21","Vals Terminal-Bench 2.1","Vals AI","knowledge","2026","State-of-the-art set of difficult terminal-based tasks",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Terminal-Bench 2.1","https://www.vals.ai/benchmarks/terminal-bench-2-1",null,"2026-08-01","registry-only",null],["benchlm-valstimehorizonksp","Vals Time Horizon Index: Kerbal Space Program","Vals AI","knowledge","2026","A Vals AI agent benchmark that gives each system five days to build and run a space program in Kerbal Space Program.",null,"higher",null,"reference",0,"review-pending","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Time Horizon Index: KSP","https://www.vals.ai/benchmarks/time_horizon_index",null,"2026-08-01","registry-only",null],["benchlm-valswebsearchindex","Vals Web Search Index","Vals AI","knowledge","2026","A Vals AI comparison of native provider search and Exa across finance-analysis and legal-research tasks.",null,"higher",null,"reference",0,"review-pending","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Web Search Index","https://www.vals.ai/benchmarks/web_search",null,"2026-08-01","registry-only",null],["benchlm-valsgpqadiamond","Vals-hosted GPQA Diamond mirror","Vals AI","knowledge","2026","Vals AI hosted GPQA Diamond view with few-shot and zero-shot chain-of-thought task splits.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Vals GPQA Diamond","https://www.vals.ai/benchmarks/gpqa",null,"2026-08-01","registry-only",null],["benchlm-valslivecodebench","Vals-hosted LiveCodeBench mirror","Vals AI","knowledge","2026","Vals AI implementation of LiveCodeBench with easy, medium, and hard task splits.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Vals LiveCodeBench","https://www.vals.ai/benchmarks/lcb",null,"2026-08-01","registry-only",null],["benchlm-valsmmlupro","Vals-hosted MMLU-Pro mirror","Vals AI","knowledge","2026","Vals AI hosted MMLU-Pro view with subject-level task splits.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Vals MMLU-Pro","https://www.vals.ai/benchmarks/mmlu_pro",null,"2026-08-01","registry-only",null],["benchlm-valsswebench","Vals-hosted SWE-bench mirror","Vals AI","knowledge","2026","Vals AI hosted SWE-bench view for solving production software engineering tasks.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Vals SWE-bench","https://www.vals.ai/benchmarks/swebench",null,"2026-08-01","registry-only",null],["benchlm-valsterminalbench2","Vals-hosted Terminal-Bench 2.0 mirror","Vals AI","knowledge","2026","Vals AI hosted Terminal-Bench 2.0 view with easy, medium, and hard task splits.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Vals Terminal-Bench 2.0","https://www.vals.ai/benchmarks/terminal-bench-2",null,"2026-08-01","registry-only",null],["benchlm-vibecodebench","Vibe Code Bench v1.1","Vals AI","coding","2026","Vals.ai benchmark for evaluating whether models can build complete web applications from natural language specifications in a production-like development environment.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Vibe Code Bench: Evaluating AI Models on End-to-End Web Application Development","https://www.vals.ai/benchmarks/vibe-code",null,"2026-08-01","registry-only",null],["benchlm-nextjsevals","AI Agent Evaluations for Next.js","Vercel","coding","2026","A Vercel benchmark for AI coding agents on Next.js code generation and migration tasks, reporting success rate, average execution time, and an AGENTS.md documentation-assisted split.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"AI Agent Evaluations | Next.js","https://nextjs.org/evals",null,"2026-08-01","registry-only",null],["benchlm-milu","Multi-task Indic Language Understanding Benchmark","Verma et al.","knowledge","2024","Culturally grounded knowledge comprehension across ten Indic languages and English.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MILU: A Multi-task Indic language understanding benchmark","https://arxiv.org/abs/2411.02538",null,"2026-08-01","registry-only",null],["benchlm-tau2airline","τ²-Bench Airline Domain","Victor Barres, Honghua Dong, Soham Ray, Xujie Si, Karthik Narasimhan","agents","2025","τ²-bench Airline tests conversational agents on airline customer-service tasks governed by domain policy and database-changing tools.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"τ²-Bench: Evaluating Conversational Agents in a Dual-Control Environment","https://arxiv.org/abs/2506.07982",null,"2026-08-01","registry-only",null],["benchlm-videomme","Video-MME","Video-MME benchmark team","multimodal","2024","A comprehensive benchmark for multimodal large language models on video understanding, covering temporal reasoning, perception, and question answering over videos.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"Video-MME benchmark","https://mme-benchmark.github.io/",null,"2026-08-01","registry-only",null],["video-mmmu","Video-MMMU","Video-MMMU","multimodal",null,"Video multimodal multi-discipline understanding benchmark stressing temporal and spatial reasoning over lecture-style video content.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"unknown","direct",null,null,"videommmu-official","production::videommmu-official","videommmu-official","Video-MMMU project","https://videommmu.github.io/",null,"2026-08-01","registry-only",null],["benchlm-lisanbench","LisanBench","voice-from-the-outer-world","reasoning","2026","A word-chain reasoning benchmark that tests planning, recall, constraint following, and vocabulary depth by asking models to extend non-repeating edit-distance-1 chains.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"LisanBench methodology","https://lisanbench.com/?tab=about",null,"2026-08-01","registry-only",null],["benchlm-voxelbench-image","VoxelBench Image-Prompt Leaderboard","VoxelBench","knowledge","2025","A live human-preference benchmark where multimodal models build voxel structures from image references and voters compare anonymous results produced from the same prompt.",null,"higher",null,"reference",0,"review-pending","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"VoxelBench leaderboard","https://voxelbench.ai/leaderboard",null,"2026-08-01","registry-only",null],["benchlm-voxelbench","VoxelBench Text-Prompt Leaderboard","VoxelBench","knowledge","2025","A live human-preference benchmark where language models turn text prompts into voxel structures and voters compare anonymous builds from the same prompt.",null,"higher",null,"reference",0,"review-pending","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"VoxelBench leaderboard","https://voxelbench.ai/leaderboard",null,"2026-08-01","registry-only",null],["benchlm-vulcanbench","VulcanBench v3","VulcanBench contributors","coding","2026","An open software-engineering benchmark built from real merged post-cutoff pull requests across Python, Rust, TypeScript, JavaScript, and Go repositories.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"VulcanBench","https://github.com/morganlinton/VulcanBench/tree/main",null,"2026-08-01","registry-only",null],["webarena","WebArena","WebArena authors","agents",null,"A benchmark of autonomous agents completing realistic tasks on self-hosted web applications.",null,"higher",null,"not-weighted",0,"active",null,null,"low","direct",null,null,null,null,null,"WebArena","https://webarena.dev/",null,"2026-07-14","registry-only",null],["webvoyager","WebVoyager","WebVoyager authors","agents",null,"End-to-end web browser agent benchmark evaluating multi-step navigation and task completion on live sites.",null,"higher",null,"not-weighted",0,"archived","percent-direct-v1",1,"medium","direct",null,null,"webvoyager-source","production::webvoyager-source","webvoyager-source","WebVoyager","https://github.com/MinorJerry/WebVoyager",null,"2026-08-01","source-qualified",null],["wildbench","WildBench","WildBench authors","instruction-following",null,"A benchmark based on challenging real-world user conversations and automated evaluation.",null,"higher",null,"not-weighted",0,"archived",null,null,"unknown","direct",null,null,null,null,null,"WildBench official repository","https://github.com/allenai/WildBench",null,"2026-08-01","registry-only",null],["benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","knowledge","2025","An expanded version of GPQA that evaluates graduate-level knowledge and reasoning capabilities across 285 disciplines, providing comprehensive coverage of academic domains.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","https://arxiv.org/abs/2502.14739",null,"2026-08-01","registry-only",null],["benchlm-programbenchepisode1","ProgramBench hidden-test pass rate after episode 1","Yang et al.","coding","2026","Program-reconstruction hidden-test pass rate after the first of five sequential long-context episodes.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"ProgramBench: Can language models rebuild programs from scratch?","https://arxiv.org/abs/2605.03546",null,"2026-08-01","registry-only",null],["benchlm-mmlupro","Massive Multitask Language Understanding Professional","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","knowledge","2024","An enhanced version of MMLU with 10 answer choices instead of 4, featuring more reasoning-focused questions that better differentiate frontier models.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","https://arxiv.org/abs/2406.01574",null,"2026-08-01","registry-only",null],["benchlm-jobbench","JobBench","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","agents","2026","An occupational agent benchmark for professional workflows that workers say they most want delegated to AI.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"JobBench: Aligning Agent Work With Human Will","https://arxiv.org/abs/2605.26329",null,"2026-08-01","registry-only",null],["benchlm-androidworld","AndroidWorld","Z.AI","agents","2026","A mobile GUI agent benchmark for completing Android app workflows and on-device tasks.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GLM-5V-Turbo","https://docs.z.ai/guides/vlm/glm-5v-turbo",null,"2026-08-01","registry-only",null],["benchlm-browsecompvl","BrowseComp-VL","Z.AI","agents","2026","A vision-language browsing benchmark for multimodal web research and tool-use workflows.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GLM-5V-Turbo","https://docs.z.ai/guides/vlm/glm-5v-turbo",null,"2026-08-01","registry-only",null],["benchlm-design2code","Design2Code","Z.AI","multimodal","2026","A multimodal coding benchmark for turning visual designs into working frontend implementations.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GLM-5V-Turbo","https://docs.z.ai/guides/vlm/glm-5v-turbo",null,"2026-08-01","registry-only",null],["benchlm-factsvlm","Facts-VLM","Z.AI","multimodal","2026","A grounded multimodal factuality benchmark for evidence-linked answer correctness.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GLM-5V-Turbo","https://docs.z.ai/guides/vlm/glm-5v-turbo",null,"2026-08-01","registry-only",null],["benchlm-flamevlmcode","Flame-VLM-Code","Z.AI","multimodal","2026","A vision-language coding benchmark for generating correct code from visual and multimodal inputs.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GLM-5V-Turbo","https://docs.z.ai/guides/vlm/glm-5v-turbo",null,"2026-08-01","registry-only",null],["benchlm-imagemining","ImageMining","Z.AI","multimodal","2026","A multimodal retrieval and extraction benchmark over image-heavy task settings.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GLM-5V-Turbo","https://docs.z.ai/guides/vlm/glm-5v-turbo",null,"2026-08-01","registry-only",null],["benchlm-mmsearch","MMSearch","Z.AI","multimodal","2026","A multimodal search benchmark for retrieval and grounded answering across mixed-media inputs.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GLM-5V-Turbo","https://docs.z.ai/guides/vlm/glm-5v-turbo",null,"2026-08-01","registry-only",null],["benchlm-mmsearchplus","MMSearch-Plus","Z.AI","multimodal","2026","A harder MMSearch variant for multimodal retrieval and grounded tool-use workflows.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GLM-5V-Turbo","https://docs.z.ai/guides/vlm/glm-5v-turbo",null,"2026-08-01","registry-only",null],["benchlm-simplevqa","SimpleVQA","Z.AI","multimodal","2026","A visual question answering benchmark focused on straightforward image-grounded understanding.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GLM-5V-Turbo","https://docs.z.ai/guides/vlm/glm-5v-turbo",null,"2026-08-01","registry-only",null],["benchlm-vstar","V*","Z.AI","multimodal","2026","A vision-centric benchmark for high-level multimodal reasoning and perception quality.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GLM-5V-Turbo","https://docs.z.ai/guides/vlm/glm-5v-turbo",null,"2026-08-01","registry-only",null],["benchlm-vision2web","Vision2Web","Z.AI","multimodal","2026","A benchmark for converting visual references into functional web implementations.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GLM-5V-Turbo","https://docs.z.ai/guides/vlm/glm-5v-turbo",null,"2026-08-01","registry-only",null],["zai-code-bench","Z.ai Code Bench","Z.AI","coding","2026-08","Z.AI in-house coding-agent benchmark covering realistic local development environments, reported at Max effort in the GLM-5.3 launch post.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"unknown","direct",null,null,"zai-glm-5-3-blog-2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3 launch blog","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","source-qualified",null],["benchlm-zclawbench","ZClawBench","Z.AI","agents","2026","A Z.AI benchmark for OpenClaw-style agent workflows spanning information search, office work, data analysis, development and operations, automation, and security.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"GLM-5-Turbo","https://docs.z.ai/guides/llm/glm-5-turbo",null,"2026-08-01","registry-only",null],["benchlm-musr","Testing the Limits of Chain-of-thought with Multistep Soft Reasoning","Zayne Sprague, Xi Ye, Kaj Bostrom, Swarat Chaudhuri, Greg Durrett","reasoning","2023","A dataset for evaluating language models on multistep soft reasoning tasks specified in natural language narratives. Tests the ability to perform complex, structured reasoning.",null,"higher",null,"reference",0,"archived","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"MuSR: Testing the Limits of Chain-of-thought with Multistep Soft Reasoning","https://arxiv.org/abs/2310.16049",null,"2026-08-01","registry-only",null],["zerobench","ZeroBench","ZeroBench","multimodal",null,"Visual reasoning stress test designed so current frontier models score near zero, measuring progress on hard multimodal puzzles.",null,"higher",null,"reference",0,"active","percent-direct-v1",1,"low","direct",null,null,"registry-zerobench","production::registry-zerobench","registry-zerobench","ZeroBench","https://zerobench.github.io/",null,"2026-08-01","registry-only",null],["benchlm-benchcadvision2codewithtools","BenchCAD Vision2Code voxel IoU with tools","Zhang et al. and Anthropic","multimodal","2026","Generates CadQuery code from multi-view renders with image inspection and code-execution tools.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"BenchCAD: A comprehensive, industry-standard benchmark for programmatic CAD","https://arxiv.org/abs/2605.10865",null,"2026-08-01","registry-only",null],["benchlm-benchcadvision2code","BenchCAD Vision2Code voxel IoU without tools","Zhang et al. and Anthropic","multimodal","2026","Generates CadQuery code from multi-view renders and scores geometric similarity by voxel intersection-over-union.",null,"higher",null,"reference",0,"active","field-relative-reference-v1",null,"unknown","direct",null,null,null,null,null,"BenchCAD: A comprehensive, industry-standard benchmark for programmatic CAD","https://arxiv.org/abs/2605.10865",null,"2026-08-01","registry-only",null]],"stableKey":"benchmarkSlug","table":"benchmark-definitions"}
